kb/concepts/ bekommt Areas: layout: fuer concept, Area-Titel aus jedem Type-Spec, Schwellen-Empfehlung im lint (schliesst #59)
Files changed: - CHANGES.md - README.md - VERSION - kb/concepts/Ambient Environment Dependency.md - kb/concepts/Anti-Cramming Heuristic.md - kb/concepts/Audit Trail.md - kb/concepts/BM25.md - kb/concepts/Bulk Operations.md - kb/concepts/CI Integration.md - kb/concepts/COLLECTION.md - kb/concepts/CPPC.md - kb/concepts/Checkpoint Audit.md - kb/concepts/Claude Code Auto Mode.md - kb/concepts/Command Round-Trip Integrity.md - kb/concepts/Confidence Scoring.md - kb/concepts/Consolidation Tiers.md - kb/concepts/Content Quality Control.md - kb/concepts/Context Isolation.md - kb/concepts/Contradiction Resolution.md - kb/concepts/Cross-platform Agent Skills.md - kb/concepts/Crystallization.md - kb/concepts/Delete Rather Than Anonymize.md - kb/concepts/Denylist over Allowlist.md - kb/concepts/Detect-Repair Asymmetry.md - kb/concepts/Diff-Reviewable Agent Edits.md - kb/concepts/Dual Licensing by File Plan.md - kb/concepts/Entity Extraction.md - kb/concepts/Episodic Memory.md - kb/concepts/Event-Driven Automation.md - kb/concepts/Filter on Ingest.md - kb/concepts/Forgetting.md - kb/concepts/Graph Traversal.md - kb/concepts/Green Suite Blind Spot.md - kb/concepts/Hooks.md - kb/concepts/Hybrid Search.md - kb/concepts/INDEX.md - kb/concepts/Implementation Spectrum.md - kb/concepts/Index Scaling.md - kb/concepts/Issue Label Scheme.md - kb/concepts/Iteration and Cost Limits.md - kb/concepts/KB Migration.md - kb/concepts/KB Stack Versioning.md - kb/concepts/Knowledge Compounding.md - kb/concepts/Knowledge Graph.md - kb/concepts/LLM Wiki Pattern.md - kb/concepts/Lint Workflow.md - kb/concepts/MCP-Leseserver.md - kb/concepts/Mass-Update Gate.md - kb/concepts/Memory Lifecycle.md - kb/concepts/Mesh Sync.md - kb/concepts/Modbus.md - kb/concepts/Multi-Agent Collaboration.md - kb/concepts/Naming Convention Conflict.md - kb/concepts/OKF Compatibility.md - kb/concepts/Optional Instance Context File.md - kb/concepts/Personalization Plane.md - kb/concepts/Privacy and Governance.md - kb/concepts/Procedural Memory.md - kb/concepts/Publish-Remote Gate.md - kb/concepts/Quality Scoring.md - kb/concepts/Quality and Self-Correction.md - kb/concepts/RAG.md - kb/concepts/Reciprocal Rank Fusion.md - kb/concepts/SSD TRIM.md - kb/concepts/Scale Ceiling.md - kb/concepts/Self-Healing.md - kb/concepts/Semantic Lint Automation.md - kb/concepts/Semantic Memory.md - kb/concepts/Session Orientation.md - kb/concepts/Shared vs Private.md - kb/concepts/Split Merge Reclassify.md - kb/concepts/Split Threshold.md - kb/concepts/Structural Enforcement over Documented Rule.md - kb/concepts/Stub Threshold.md - kb/concepts/Supersession.md - kb/concepts/Three-Layer Architecture.md - kb/concepts/Token Economics.md - kb/concepts/Typed Relationships.md - kb/concepts/User Management.md - kb/concepts/Vector Search.md - kb/concepts/Work Coordination.md - kb/concepts/Workflow Extraction.md - kb/concepts/Workflow Orchestration.md - kb/concepts/Working Memory.md - kb/concepts/Write-Once Frontmatter Fields.md - kb/concepts/architectures/Consolidation Tiers.md - kb/concepts/architectures/Context Isolation.md - kb/concepts/architectures/Cross-platform Agent Skills.md - kb/concepts/architectures/Episodic Memory.md - kb/concepts/architectures/Hybrid Search.md - kb/concepts/architectures/Implementation Spectrum.md - kb/concepts/architectures/Knowledge Graph.md - kb/concepts/architectures/LLM Wiki Pattern.md - kb/concepts/architectures/MCP-Leseserver.md - kb/concepts/architectures/Memory Lifecycle.md - kb/concepts/architectures/OKF Compatibility.md - kb/concepts/architectures/Optional Instance Context File.md - kb/concepts/architectures/Personalization Plane.md - kb/concepts/architectures/Procedural Memory.md - kb/concepts/architectures/RAG.md - kb/concepts/architectures/Scale Ceiling.md - kb/concepts/architectures/Semantic Memory.md - kb/concepts/architectures/Three-Layer Architecture.md - kb/concepts/architectures/Token Economics.md - kb/concepts/architectures/Working Memory.md - kb/concepts/decisions/Delete Rather Than Anonymize.md - kb/concepts/decisions/Denylist over Allowlist.md - kb/concepts/decisions/Diff-Reviewable Agent Edits.md - kb/concepts/decisions/Dual Licensing by File Plan.md - kb/concepts/decisions/Issue Label Scheme.md - kb/concepts/decisions/KB Stack Versioning.md - kb/concepts/decisions/Structural Enforcement over Documented Rule.md - kb/concepts/patterns/Audit Trail.md - kb/concepts/patterns/BM25.md - kb/concepts/patterns/Command Round-Trip Integrity.md - kb/concepts/patterns/Confidence Scoring.md - kb/concepts/patterns/Contradiction Resolution.md - kb/concepts/patterns/Entity Extraction.md - kb/concepts/patterns/Filter on Ingest.md - kb/concepts/patterns/Forgetting.md - kb/concepts/patterns/Graph Traversal.md - kb/concepts/patterns/Mesh Sync.md - kb/concepts/patterns/Quality Scoring.md - kb/concepts/patterns/Reciprocal Rank Fusion.md - kb/concepts/patterns/Self-Healing.md - kb/concepts/patterns/Shared vs Private.md - kb/concepts/patterns/Typed Relationships.md - kb/concepts/patterns/Vector Search.md - kb/concepts/patterns/Work Coordination.md - kb/concepts/problems/Ambient Environment Dependency.md - kb/concepts/problems/Detect-Repair Asymmetry.md - kb/concepts/problems/Green Suite Blind Spot.md - kb/concepts/problems/Naming Convention Conflict.md - kb/concepts/problems/Write-Once Frontmatter Fields.md - kb/concepts/protocols/CPPC.md - kb/concepts/protocols/Modbus.md - kb/concepts/protocols/SSD TRIM.md - kb/concepts/workflows/Anti-Cramming Heuristic.md - kb/concepts/workflows/Bulk Operations.md - kb/concepts/workflows/CI Integration.md - kb/concepts/workflows/Checkpoint Audit.md - kb/concepts/workflows/Claude Code Auto Mode.md - kb/concepts/workflows/Content Quality Control.md - kb/concepts/workflows/Crystallization.md - kb/concepts/workflows/Event-Driven Automation.md - kb/concepts/workflows/Hooks.md - kb/concepts/workflows/Index Scaling.md - kb/concepts/workflows/Iteration and Cost Limits.md - kb/concepts/workflows/KB Migration.md - kb/concepts/workflows/Knowledge Compounding.md - kb/concepts/workflows/Lint Workflow.md - kb/concepts/workflows/Mass-Update Gate.md - kb/concepts/workflows/Multi-Agent Collaboration.md - kb/concepts/workflows/Privacy and Governance.md - kb/concepts/workflows/Publish-Remote Gate.md - kb/concepts/workflows/Quality and Self-Correction.md - kb/concepts/workflows/Semantic Lint Automation.md - kb/concepts/workflows/Session Orientation.md - kb/concepts/workflows/Split Merge Reclassify.md - kb/concepts/workflows/Split Threshold.md - kb/concepts/workflows/Stub Threshold.md - kb/concepts/workflows/Supersession.md - kb/concepts/workflows/User Management.md - kb/concepts/workflows/Workflow Extraction.md - kb/concepts/workflows/Workflow Orchestration.md - kb/index.md - kb/log.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/catalog.py - tools/chemenu/commands/index_build.py - tools/chemenu/lint_core.py - tools/chemenu/tests/conftest.py - tools/chemenu/tests/test_cite_cmd.py - tools/chemenu/tests/test_git_publish.py - tools/chemenu/tests/test_index_build.py - tools/chemenu/tests/test_lint.py - tools/chemenu/tests/test_new_page.py - tools/chemenu/tests/test_provenance.py - tools/chemenu/tests/test_type_resolver.py - tools/chemenu/tests/test_xref.py - types/concept.md - types/type-spec.md
This commit is contained in:
+1
-1
@@ -51,7 +51,7 @@ tools/wikitool <command> --help
|
||||
| `index rebuild [--dry-run]` | Regenerate the catalog from every page's frontmatter: `kb/index.md` becomes a map (statistics, one row per collection and per area, links to the shards) and the page tables are written to a generated `INDEX.md` in each collection. An area past 50 rows gets its own shard. Stale shards from removed collections/areas are deleted in the same pass |
|
||||
| `log append --op ingest\|query\|lint\|create\|update\|delete\|rename\|move --title "..." [--body "..."\|--body-file path]` | Append a formatted entry to `kb/log.md` |
|
||||
| `log status` | Read-only: count `ingest` entries logged since the last `lint` entry - the deterministic trigger behind the Maintenance Schedule's "every 10 sources" full-lint cadence |
|
||||
| `lint [--json] [--markdown out.md] [--full] [--fail-on-error]` | Structural + provenance checks: broken wikilinks, dangling frontmatter references, orphan pages, index drift, schema gaps, duplicate titles, title mismatches, pages nested more than one directory below their collection (hard - the generated catalog folds these into their area silently rather than merely reading it, see #57), uncovered raw files, broken `raw_files:` refs, raw files claimed by more than one source page, unmarked provenance, citation/frontmatter drift, unbalanced generated-region markers, edges whose label is missing or not authorised by the source collection's `outbound:` (both hard once `kb_version` has reached the release that introduced labelled edges - advisory below it, so a corpus mid-migration is not refused by the check measuring it), `see-also` edges whose reverse direction already carries a specific label (advisory only - redundant rather than wrong, and never migration-gated, since no version turns the redundancy into an error), quote-limit overages (>2 blockquoted lines/page, advisory only). Prints only the sections that found something and always writes the full report to `reports/Lint Report <date>.md` (or `--markdown`), naming the path - `--full` prints everything, `--json` prints the findings and writes nothing |
|
||||
| `lint [--json] [--markdown out.md] [--full] [--fail-on-error]` | Structural + provenance checks: broken wikilinks, dangling frontmatter references, orphan pages, index drift, schema gaps, duplicate titles, title mismatches, pages nested more than one directory below their collection (hard - the generated catalog folds these into their area silently rather than merely reading it, see #57), uncovered raw files, broken `raw_files:` refs, raw files claimed by more than one source page, unmarked provenance, citation/frontmatter drift, unbalanced generated-region markers, edges whose label is missing or not authorised by the source collection's `outbound:` (both hard once `kb_version` has reached the release that introduced labelled edges - advisory below it, so a corpus mid-migration is not refused by the check measuring it), `see-also` edges whose reverse direction already carries a specific label (advisory only - redundant rather than wrong, and never migration-gated, since no version turns the redundancy into an error), a collection past the catalog's per-area shard threshold that has no areas to shard (advisory only - sharding is automatic but per *area*, so a collection nobody gave areas keeps one table however large it grows, #59; reported with the split its subtype field would produce, and only when that split puts every resulting area at or under the threshold, so a lopsided or small collection stays silent), quote-limit overages (>2 blockquoted lines/page, advisory only). Prints only the sections that found something and always writes the full report to `reports/Lint Report <date>.md` (or `--markdown`), naming the path - `--full` prints everything, `--json` prints the findings and writes nothing |
|
||||
| `search ["<text>"] [--field <predicate> ...] [--kind/--subtype/--collection/--tag <v>] [--regex] [--limit N] [--sort [-]<field>] [--backend <name>] [--matches] [--json]` | Find pages in `kb/` without reading the index. Text search runs through a pluggable backend (`rg` today); `--field` predicates are evaluated on frontmatter - `f=v`, `f~substring`, `'f>=v'`, `'f:*'` (present), `'!f'` (absent), repeatable and ANDed. With no text this is a pure structured query. Results carry kind/summary/confidence so a hit can be judged without opening the page. A page whose frontmatter does not parse can match no positive predicate, so it is **named** rather than dropped: `--json` always carries an `unreadable` list of `{path, reason}` (usually empty), and the table form writes the same lines to stderr. `--regex` is applied by `rg` alone, whose engine is linear; the ranking boosts for title and summary are literal-containment only, so a non-literal pattern is ranked by match count. `rg` is killed after 30 s and reported as a failure. Read-only, and **exempt from the Iteration Budget Gate** |
|
||||
| `confidence decay [--apply]` | Recompute every page's derived `confidence` as `confidence_base * (1 - 0.01/month)`, floored at 0.2; dry-run by default |
|
||||
| `confidence init-base [--apply]` | One-time backfill: set `confidence_base` from the current `confidence` on pages that predate the derived-confidence model |
|
||||
|
||||
+3
-2
@@ -45,6 +45,7 @@ tools/
|
||||
conventions.py kb/CONVENTIONS.md: what this instance decided about authoring, as opposed to what the stack enforces
|
||||
ownership.py the stack-vs-instance boundary under a content stage - one predicate, read by `dist_cmd.py` and `commands/upstream_cmd.py` so the two cannot answer it differently
|
||||
type_resolver.py type-spec loading and schema resolution
|
||||
catalog.py how the corpus groups into collections and areas, and the shard threshold - with no CLI attached
|
||||
lint_core.py the lint checks and the report, with no CLI attached
|
||||
types_core.py type-spec listing/description, with no CLI attached
|
||||
markdown_code.py masks code spans/fences so a page may show wiki notation, not only use it
|
||||
@@ -57,8 +58,8 @@ tools/
|
||||
```
|
||||
|
||||
**Two consumers, one core.** The CLI is not the only caller any more. The cores
|
||||
(`search/service.py`, `lint_core.py`, `types_core.py`) hold what decides an
|
||||
answer and import no `typer` and no `rich`; the modules under `commands/` turn
|
||||
(`search/service.py`, `lint_core.py`, `types_core.py`, `catalog.py`) hold what
|
||||
decides an answer and import no `typer` and no `rich`; the modules under `commands/` turn
|
||||
those values into terminal output and those exceptions into exit codes.
|
||||
`api.Corpus` is the in-process entry point over the same functions - it takes a
|
||||
corpus root, returns exactly the structures the `--json` forms print, and
|
||||
|
||||
@@ -0,0 +1,140 @@
|
||||
"""How the corpus groups into collections and areas, with no CLI attached.
|
||||
|
||||
Split out of `commands/index_build.py` for the reason `lint_core.py` gives at
|
||||
the top of itself: this is a pure function over a corpus directory, and it was
|
||||
sitting in a module that imports `typer` and `rich`. `lint` needs the same
|
||||
grouping - it is what answers "does this collection have areas, and is it over
|
||||
the threshold?" (Gitea #59) - and `chemenu.api`, the read surface, may not
|
||||
reach a command module at all. Importing it from there would have pulled the
|
||||
whole CLI head in behind it.
|
||||
|
||||
So the split runs along the same line as lint's: everything that decides *how
|
||||
the corpus is shaped* lives here; everything that decides *what the catalog
|
||||
looks like* - the tables, the map, the shard files - stays in
|
||||
`commands/index_build.py`, which imports from here.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from chemenu.kb_collections import iter_kb_collections
|
||||
from chemenu.page import Page
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
# Rows per area before it is split into its own shard. From the wiki's own
|
||||
# `Index Scaling` page ("split table sections at >50 entries"), kept as a plain
|
||||
# number so growth is handled by arithmetic rather than by a judgment call.
|
||||
SHARD_THRESHOLD = 50
|
||||
|
||||
# Display title for pages sitting directly in a collection root rather than in
|
||||
# an area subdirectory.
|
||||
UNGROUPED_TITLE = "All"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Area:
|
||||
"""One grouping inside a collection: a subdirectory, or the collection root
|
||||
for pages that sit directly in it."""
|
||||
|
||||
name: str
|
||||
title: str
|
||||
pages: list[Page] = field(default_factory=list)
|
||||
own_shard: bool = False
|
||||
|
||||
@property
|
||||
def count(self) -> int:
|
||||
return len(self.pages)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Collection:
|
||||
name: str
|
||||
areas: list[Area] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def count(self) -> int:
|
||||
return sum(area.count for area in self.areas)
|
||||
|
||||
|
||||
def area_titles() -> dict[str, dict[str, str]]:
|
||||
"""Display titles per collection: `{collection: {area_dir: title}}`, taken
|
||||
from each type-spec's own `layout:` rather than a hardcoded map - so a new
|
||||
subtype names its own section by adding a type-spec, with no code change.
|
||||
|
||||
Every type-spec is read, not just `entity`'s. That hardcoding was the
|
||||
asymmetry behind Gitea #59: the axis a collection splits along is declared
|
||||
in `layout:`, and a second type declaring one would have had its areas
|
||||
titled by `.title()` on the directory name while entity's got their real
|
||||
names.
|
||||
|
||||
Keyed by collection rather than by directory name alone, because two types
|
||||
writing into two collections may legitimately use the same area name for
|
||||
different things (`kb/entities/tools/` and a hypothetical
|
||||
`kb/concepts/tools/`); a flat map would hand the second one the first's
|
||||
title. The collection key is the type's `base_dir:`, which is what put the
|
||||
page in that directory to begin with.
|
||||
"""
|
||||
titles: dict[str, dict[str, str]] = {}
|
||||
for type_path, _frontmatter in resolver.list_type_specs():
|
||||
try:
|
||||
if resolver.get_root(type_path) != "kb":
|
||||
continue
|
||||
base_dir = resolver.get_base_dir(type_path)
|
||||
layout = resolver.get_layout(type_path)
|
||||
except (ValueError, OSError):
|
||||
continue
|
||||
if not base_dir or not layout:
|
||||
continue
|
||||
per_collection = titles.setdefault(str(base_dir).strip("/"), {})
|
||||
for key, spec in layout.items():
|
||||
per_collection.setdefault(spec.get("dir", key), spec.get("title", str(key).title()))
|
||||
return titles
|
||||
|
||||
|
||||
def group_pages(kb_dir: Path, pages: dict[str, Page]) -> list[Collection]:
|
||||
"""Group pages by their physical location: collection directory, then area
|
||||
subdirectory.
|
||||
|
||||
Location rather than `kind` because a shard lives in the directory it
|
||||
describes, and the two agree by construction: a type-spec's `base_dir:` is
|
||||
what put the page there.
|
||||
"""
|
||||
titles = area_titles()
|
||||
grouped: dict[str, dict[str, Area]] = {}
|
||||
|
||||
# Seed from the collections that exist on disk, not only from the ones that
|
||||
# happen to hold pages: an empty collection is a real (if unfilled) part of
|
||||
# the wiki, and dropping it from the map would hide it from every reader.
|
||||
for collection_dir in iter_kb_collections(kb_dir):
|
||||
grouped.setdefault(collection_dir.name, {})
|
||||
|
||||
for page in sorted(pages.values(), key=lambda p: p.title.lower()):
|
||||
try:
|
||||
parts = page.path.relative_to(kb_dir).parts
|
||||
except ValueError: # pragma: no cover - pages always live under kb_dir
|
||||
continue
|
||||
if len(parts) < 2:
|
||||
collection_name, area_name = "(kb root)", ""
|
||||
else:
|
||||
collection_name = parts[0]
|
||||
area_name = parts[1] if len(parts) > 2 else ""
|
||||
areas = grouped.setdefault(collection_name, {})
|
||||
area = areas.get(area_name)
|
||||
if area is None:
|
||||
title = (
|
||||
titles.get(collection_name, {}).get(area_name, area_name.title())
|
||||
if area_name
|
||||
else UNGROUPED_TITLE
|
||||
)
|
||||
area = Area(name=area_name, title=title)
|
||||
areas[area_name] = area
|
||||
area.pages.append(page)
|
||||
|
||||
collections = []
|
||||
for name in sorted(grouped):
|
||||
ordered = sorted(grouped[name].values(), key=lambda a: (a.name == "", a.title.lower()))
|
||||
for area in ordered:
|
||||
area.own_shard = bool(area.name) and area.count > SHARD_THRESHOLD
|
||||
collections.append(Collection(name=name, areas=ordered))
|
||||
return collections
|
||||
@@ -18,18 +18,16 @@ The map stays small enough to browse; `wikitool search` answers everything else.
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
import typer
|
||||
|
||||
from chemenu import config
|
||||
from chemenu.catalog import SHARD_THRESHOLD, Area, Collection, group_pages
|
||||
from chemenu.commands._util import rel_path, success
|
||||
from chemenu.kb_collections import iter_kb_collections
|
||||
from chemenu.page import Page
|
||||
from chemenu.kb_scan import GENERATED_INDEX, find_nested_pages, load_kb_pages
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
app = typer.Typer(help="Manage the generated wiki catalog (kb/index.md + per-collection INDEX.md).")
|
||||
|
||||
@@ -38,15 +36,6 @@ TABLE_SEP = "|------|------|---------|----------------|"
|
||||
|
||||
SUMMARY_HEADINGS = ("Description", "Definition", "Summary")
|
||||
|
||||
# Rows per area before it is split into its own shard. From the wiki's own
|
||||
# `Index Scaling` page ("split table sections at >50 entries"), kept as a plain
|
||||
# number so growth is handled by arithmetic rather than by a judgment call.
|
||||
SHARD_THRESHOLD = 50
|
||||
|
||||
# Display title for pages sitting directly in a collection root rather than in
|
||||
# an area subdirectory.
|
||||
UNGROUPED_TITLE = "All"
|
||||
|
||||
DO_NOT_EDIT = "<!-- Generated by `wikitool index rebuild`. Do not hand-edit. -->"
|
||||
|
||||
|
||||
@@ -85,86 +74,17 @@ def _table(pages: list[Page]) -> list[str]:
|
||||
|
||||
|
||||
def _anchor(title: str) -> str:
|
||||
"""GitHub-style heading anchor, so the map can deep-link into a shard."""
|
||||
slug = re.sub(r"[^a-z0-9\s-]", "", title.lower())
|
||||
return re.sub(r"\s+", "-", slug.strip())
|
||||
"""GitHub-style heading anchor, so the map can deep-link into a shard.
|
||||
|
||||
|
||||
@dataclass
|
||||
class Area:
|
||||
"""One grouping inside a collection: a subdirectory, or the collection root
|
||||
for pages that sit directly in it."""
|
||||
|
||||
name: str
|
||||
title: str
|
||||
pages: list[Page] = field(default_factory=list)
|
||||
own_shard: bool = False
|
||||
|
||||
@property
|
||||
def count(self) -> int:
|
||||
return len(self.pages)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Collection:
|
||||
name: str
|
||||
areas: list[Area] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def count(self) -> int:
|
||||
return sum(area.count for area in self.areas)
|
||||
|
||||
|
||||
def _area_titles() -> dict[str, str]:
|
||||
"""Display titles for entity areas, taken from the entity type-spec's own
|
||||
`layout:` rather than a hardcoded map - so a new subtype names its own
|
||||
section by adding a type-spec, with no code change."""
|
||||
layout = resolver.get_layout(resolver.find_type_by_name("entity")) or {}
|
||||
return {spec.get("dir", key): spec.get("title", key.title()) for key, spec in layout.items()}
|
||||
|
||||
|
||||
def group_pages(kb_dir: Path, pages: dict[str, Page]) -> list[Collection]:
|
||||
"""Group pages by their physical location: collection directory, then area
|
||||
subdirectory.
|
||||
|
||||
Location rather than `kind` because a shard lives in the directory it
|
||||
describes, and the two agree by construction: a type-spec's `base_dir:` is
|
||||
what put the page there.
|
||||
`\\w` rather than `a-z0-9`, which is not cosmetic: an area title follows the
|
||||
KB language, and the first non-English one (`Abläufe`) had its umlaut
|
||||
*deleted* rather than kept, so the map linked at `#ablufe` and the anchor it
|
||||
was aiming at was `#abläufe`. Every deep link into a shard whose title
|
||||
carries a non-ASCII letter was silently dead. Nothing surfaced it while the
|
||||
only areas were entity ones, whose titles happen to be ASCII throughout.
|
||||
"""
|
||||
titles = _area_titles()
|
||||
grouped: dict[str, dict[str, Area]] = {}
|
||||
|
||||
# Seed from the collections that exist on disk, not only from the ones that
|
||||
# happen to hold pages: an empty collection is a real (if unfilled) part of
|
||||
# the wiki, and dropping it from the map would hide it from every reader.
|
||||
for collection_dir in iter_kb_collections(kb_dir):
|
||||
grouped.setdefault(collection_dir.name, {})
|
||||
|
||||
for page in sorted(pages.values(), key=lambda p: p.title.lower()):
|
||||
try:
|
||||
parts = page.path.relative_to(kb_dir).parts
|
||||
except ValueError: # pragma: no cover - pages always live under kb_dir
|
||||
continue
|
||||
if len(parts) < 2:
|
||||
collection_name, area_name = "(kb root)", ""
|
||||
else:
|
||||
collection_name = parts[0]
|
||||
area_name = parts[1] if len(parts) > 2 else ""
|
||||
areas = grouped.setdefault(collection_name, {})
|
||||
area = areas.get(area_name)
|
||||
if area is None:
|
||||
title = titles.get(area_name, area_name.title()) if area_name else UNGROUPED_TITLE
|
||||
area = Area(name=area_name, title=title)
|
||||
areas[area_name] = area
|
||||
area.pages.append(page)
|
||||
|
||||
collections = []
|
||||
for name in sorted(grouped):
|
||||
ordered = sorted(grouped[name].values(), key=lambda a: (a.name == "", a.title.lower()))
|
||||
for area in ordered:
|
||||
area.own_shard = bool(area.name) and area.count > SHARD_THRESHOLD
|
||||
collections.append(Collection(name=name, areas=ordered))
|
||||
return collections
|
||||
slug = re.sub(r"[^\w\s-]", "", title.lower(), flags=re.UNICODE)
|
||||
return re.sub(r"\s+", "-", slug.strip())
|
||||
|
||||
|
||||
def build_area_shard(area: Area) -> str:
|
||||
|
||||
@@ -16,7 +16,10 @@ from __future__ import annotations
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
from collections import Counter
|
||||
|
||||
from chemenu import blocks, config, kb_collections, links
|
||||
from chemenu.catalog import SHARD_THRESHOLD, group_pages
|
||||
from chemenu.frontmatter_io import frontmatter_error
|
||||
from chemenu.markdown_code import strip_code_spans
|
||||
from chemenu.provenance import broken_raw_refs as find_broken_raw_refs
|
||||
@@ -141,6 +144,88 @@ def nested_pages(kb_dir: Path, pages: dict[str, Page]) -> list[dict]:
|
||||
]
|
||||
|
||||
|
||||
def unsharded_collections(kb_dir: Path, pages: dict[str, Page]) -> list[dict]:
|
||||
"""Collections past the catalog's shard threshold that have no areas to
|
||||
shard, together with the subtype split that would give them some.
|
||||
|
||||
Sharding is already automatic, and it is per *area*: `index rebuild` hands
|
||||
an area over `SHARD_THRESHOLD` rows its own `INDEX.md`. Creating an area is
|
||||
not automatic and nothing ever asked for one - so a collection that never
|
||||
grew any keeps its whole catalog in a single table, past the threshold,
|
||||
forever. The threshold is then not a threshold but a dead value (Gitea
|
||||
#59), and this is the only check that can notice: an ingest sees one
|
||||
source and cannot see a collection's size, while `lint` sees the corpus and
|
||||
runs every 10 sources anyway.
|
||||
|
||||
**A recommendation, not a failure** (it is deliberately absent from
|
||||
`HARD_ERROR_KEYS`), and narrow enough to stay one: it fires only where the
|
||||
split actually helps - every area it would create, the ungrouped remainder
|
||||
included, lands at or under the threshold. That self-limits in both
|
||||
directions. A collection under the threshold never fires, so a small
|
||||
`kb/comparisons/` is not permanently in the report; and a collection whose
|
||||
subtype values are lopsided (25 of 29 `source_type: notes`) does not fire
|
||||
either, because splitting it would produce one area over the threshold and
|
||||
a handful of splinters. What is left is a finding that appears when a
|
||||
collection grows into it and is silent when it does not.
|
||||
"""
|
||||
findings: list[dict] = []
|
||||
for collection in group_pages(kb_dir, pages):
|
||||
if collection.count <= SHARD_THRESHOLD:
|
||||
continue
|
||||
# An area already exists, so the collection has been split once and
|
||||
# `index rebuild` shards whatever outgrows the threshold from here.
|
||||
# A page still sitting in the root is `misplaced_pages`' finding, not
|
||||
# this one.
|
||||
if any(area.name for area in collection.areas):
|
||||
continue
|
||||
|
||||
counts: Counter[str] = Counter()
|
||||
fields: set[str] = set()
|
||||
# Whether every type writing here already declares the `layout:` that
|
||||
# turns the subtype into a directory. It decides which half of the fix
|
||||
# is still owed: without it there is nothing for `move` to compute a
|
||||
# destination from, with it the move is all that is left.
|
||||
layouts: set[bool] = set()
|
||||
for area in collection.areas:
|
||||
for page in area.pages:
|
||||
type_path = page.frontmatter.get("type")
|
||||
if not type_path:
|
||||
continue
|
||||
try:
|
||||
field = resolver.get_subtype_field(type_path, page.path)
|
||||
layout = resolver.get_layout(type_path, page.path)
|
||||
except ValueError:
|
||||
continue
|
||||
value = page.frontmatter.get(field) if field else None
|
||||
if not value:
|
||||
continue
|
||||
counts[str(value)] += 1
|
||||
fields.add(str(field))
|
||||
layouts.add(bool(layout))
|
||||
|
||||
if not counts:
|
||||
continue
|
||||
# The pages the subtype cannot place stay in the collection root, so
|
||||
# they are an area of their own for the purpose of this test.
|
||||
unplaced = collection.count - sum(counts.values())
|
||||
if max([*counts.values(), unplaced]) > SHARD_THRESHOLD:
|
||||
continue
|
||||
|
||||
findings.append(
|
||||
{
|
||||
"collection": collection.name,
|
||||
"count": collection.count,
|
||||
"field": ", ".join(sorted(fields)),
|
||||
"layout_declared": layouts == {True},
|
||||
"distribution": [
|
||||
{"value": value, "count": count}
|
||||
for value, count in sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
|
||||
],
|
||||
}
|
||||
)
|
||||
return findings
|
||||
|
||||
|
||||
def run_lint(kb_dir: Path) -> dict:
|
||||
pages = load_kb_pages(kb_dir)
|
||||
duplicate_titles = find_duplicate_title_paths(kb_dir, config.ROOT)
|
||||
@@ -388,6 +473,7 @@ def run_lint(kb_dir: Path) -> dict:
|
||||
"duplicate_titles": duplicate_titles,
|
||||
"misplaced_pages": misplaced,
|
||||
"nested_pages": nested,
|
||||
"unsharded_collections": unsharded_collections(kb_dir, pages),
|
||||
"uncovered_raw_files": find_uncovered_raw_files(config.RAW_DIR, pages),
|
||||
"broken_raw_refs": find_broken_raw_refs(pages),
|
||||
"duplicate_raw_file_owners": find_duplicate_raw_file_owners(pages),
|
||||
@@ -464,6 +550,23 @@ def render_markdown(report: dict) -> str:
|
||||
"the catalog folds this into its area silently; `wikitool move --reconcile` fixes it "
|
||||
"when the page's type resolves to a shallower directory, otherwise move it up by hand",
|
||||
)
|
||||
_section(
|
||||
lines, f"Collections Past the Shard Threshold (>{SHARD_THRESHOLD}) With No Areas "
|
||||
"- recommendation, not an error",
|
||||
report.get("unsharded_collections", []),
|
||||
lambda i: f"`kb/{i['collection']}/` holds {i['count']} pages in a single table and has no "
|
||||
f"areas, so the per-area shard threshold never fires. Splitting on `{i['field']}` would "
|
||||
"give: "
|
||||
+ ", ".join(f"{d['value']} {d['count']}" for d in i["distribution"])
|
||||
+ f" - all at or under {SHARD_THRESHOLD}. "
|
||||
+ (
|
||||
"The type-spec already declares the `layout:` for those values, so "
|
||||
"`wikitool move --reconcile` and `wikitool index rebuild` are the whole fix"
|
||||
if i.get("layout_declared")
|
||||
else "Declare a `layout:` for those values in the type-spec, then "
|
||||
"`wikitool move --reconcile` and `wikitool index rebuild`"
|
||||
),
|
||||
)
|
||||
_section(
|
||||
lines, "Uncovered Raw Files (no source page)", report["uncovered_raw_files"],
|
||||
lambda i: f"`{i}`",
|
||||
@@ -617,6 +720,14 @@ def default_report_path(report: dict) -> Path:
|
||||
# and there is no version at which "not under the computed directory" becomes
|
||||
# wrong - only `wikitool move` someone does or does not get to run.
|
||||
#
|
||||
# `unsharded_collections` is advisory by construction rather than by tolerance:
|
||||
# it does not describe anything that is wrong, only a collection that has grown
|
||||
# past the size at which areas start paying for themselves. Whether to split it
|
||||
# is an authoring decision about how the corpus is organised - the tool can see
|
||||
# that the split would work and say so, and that is the whole of its authority.
|
||||
# Failing on it would also make `lint` red on a corpus that is entirely
|
||||
# self-consistent, which is the state the recommendation is asking to improve.
|
||||
#
|
||||
# `malformed_edges` and `unbalanced_markers` are hard from the start: neither
|
||||
# describes an unconverted page, only a broken one.
|
||||
#
|
||||
|
||||
@@ -254,7 +254,7 @@ def kb_dir(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
|
||||
kb = tmp_path / "kb"
|
||||
for sub in ("entities/projects", "entities/systems", "entities/tools",
|
||||
"entities/technologies", "entities/people",
|
||||
"concepts", "sources", "comparisons"):
|
||||
"concepts/protocols", "sources", "comparisons"):
|
||||
(kb / sub).mkdir(parents=True)
|
||||
# The contracts carry a real declaration, because three things now read one:
|
||||
# `docs verify` checks `profile:`/`required_by_stack:`, and `xref add` asks
|
||||
@@ -302,7 +302,7 @@ def kb_dir(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path:
|
||||
"\n# gdeploy\n\n## Description\n\nDeploy tool.\n",
|
||||
)
|
||||
write_page(
|
||||
kb / "concepts/Modbus.md",
|
||||
kb / "concepts/protocols/Modbus.md",
|
||||
{
|
||||
"type": "types/concept.md", "concept_type": "protocol",
|
||||
"tags": [], "created": "2026-07-25", "modified": "2026-07-25",
|
||||
|
||||
@@ -54,7 +54,7 @@ def test_cite_add_command_writes_definition_and_prints_marker(kb_dir, raw_dir, m
|
||||
marker_id = cite_id("Source - Aurora")
|
||||
assert f"[^{marker_id}]" in result.output
|
||||
|
||||
fm, body = read_page(kb_dir / "concepts/Modbus.md")
|
||||
fm, body = read_page(kb_dir / "concepts/protocols/Modbus.md")
|
||||
assert "Source - Aurora" in fm["sources"]
|
||||
assert f"[^{marker_id}]: [[Source - Aurora]]" in body
|
||||
|
||||
@@ -194,11 +194,11 @@ def test_cite_sync_command_over_kb(kb_dir, raw_dir, monkeypatch):
|
||||
assert add_result.exit_code == 0, add_result.output
|
||||
marker_id = cite_id("Source - Aurora")
|
||||
|
||||
fm, body = read_page(kb_dir / "concepts/Modbus.md")
|
||||
fm, body = read_page(kb_dir / "concepts/protocols/Modbus.md")
|
||||
body = body.replace("Industrial protocol.", f"Industrial protocol [^{marker_id}].")
|
||||
from chemenu.frontmatter_io import write_page
|
||||
|
||||
write_page(kb_dir / "concepts/Modbus.md", fm, body)
|
||||
write_page(kb_dir / "concepts/protocols/Modbus.md", fm, body)
|
||||
|
||||
result = runner.invoke(app, ["cite", "sync", "--all", "--dry-run"])
|
||||
assert result.exit_code == 0, result.output
|
||||
|
||||
@@ -494,7 +494,7 @@ def test_generated_files_are_recognised_wherever_they_sit():
|
||||
assert is_generated("kb/provenance.md")
|
||||
assert is_generated("kb/concepts/INDEX.md")
|
||||
assert is_generated("kb/entities/tools/INDEX.md")
|
||||
assert not is_generated("kb/concepts/Modbus.md")
|
||||
assert not is_generated("kb/concepts/protocols/Modbus.md")
|
||||
|
||||
|
||||
def test_paths_land_in_the_group_a_reviewer_expects():
|
||||
|
||||
@@ -15,14 +15,14 @@ from chemenu.kb_scan import GENERATED_INDEX, iter_kb_pages
|
||||
from chemenu.type_resolver import resolver
|
||||
|
||||
|
||||
def _area_title(subtype: str) -> str:
|
||||
"""The display title `index rebuild` will use for an entity subtype.
|
||||
def _area_title(subtype: str, type_path: str = "types/entity.md") -> str:
|
||||
"""The display title `index rebuild` will use for a subtype's area.
|
||||
|
||||
Read from the type-spec rather than written out, because these titles follow
|
||||
the KB language: hard-coding them made translating the wiki fail tests that
|
||||
are not about wording at all.
|
||||
"""
|
||||
return resolver.get_layout("types/entity.md")[subtype]["title"]
|
||||
return resolver.get_layout(type_path)[subtype]["title"]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
@@ -116,6 +116,28 @@ def test_area_titles_come_from_the_entity_type_spec_layout(plan, kb_dir):
|
||||
assert f"## {_area_title('tool')}" in entities
|
||||
|
||||
|
||||
def test_anchor_keeps_non_ascii_letters():
|
||||
"""An area title follows the KB language, so it may carry a letter outside
|
||||
`a-z`. Deleting it - which is what the old `[^a-z0-9\\s-]` did - produced a
|
||||
map link (`#ablufe`) that pointed at no heading in the shard it named."""
|
||||
assert _anchor("Abläufe") == "abläufe"
|
||||
assert _anchor("Größere Muster") == "größere-muster"
|
||||
# Punctuation is still dropped and any run of whitespace still collapses to
|
||||
# a single hyphen.
|
||||
assert _anchor("Tools & Utilities (v2)") == "tools-utilities-v2"
|
||||
|
||||
|
||||
def test_area_titles_are_read_from_every_type_spec_not_only_entity(plan, kb_dir):
|
||||
"""Gitea #59: the title lookup used to resolve `entity` by name and read
|
||||
only its `layout:`, so a second type declaring one got `.title()` on its
|
||||
directory name (`Protocols`) instead of the title it declared. The concept
|
||||
areas are the first case; nothing about them is special."""
|
||||
concepts = _shard(plan, kb_dir, "concepts")
|
||||
declared = _area_title("protocol", "types/concept.md")
|
||||
assert f"## {declared}" in concepts
|
||||
assert "## Protocols" not in concepts
|
||||
|
||||
|
||||
def test_summary_prefers_frontmatter_then_falls_back_to_body(plan, kb_dir):
|
||||
entities = _shard(plan, kb_dir, "entities")
|
||||
assert "Server hosting DocStore with ZFS storage" in entities # frontmatter
|
||||
|
||||
@@ -236,9 +236,133 @@ def test_lint_is_silent_about_pages_directly_in_an_area(kb_dir):
|
||||
assert run_lint(kb_dir)["nested_pages"] == []
|
||||
|
||||
|
||||
# --- collections that outgrew the shard threshold without areas (Gitea #59) ---
|
||||
|
||||
|
||||
def _fill_collection(kb_dir, collection: str, type_path: str, field: str, distribution: dict):
|
||||
"""Write pages flat into `kb/<collection>/`, `distribution` many per subtype
|
||||
value - the shape a collection is in when nobody ever created an area.
|
||||
|
||||
The precondition is established rather than assumed: the fixture corpus
|
||||
places its concept page in an area (as the real one does now), and a single
|
||||
area is enough to make this finding stand down.
|
||||
"""
|
||||
for existing in sorted((kb_dir / collection).rglob("*.md")):
|
||||
if existing.parent != kb_dir / collection:
|
||||
existing.rename(kb_dir / collection / existing.name)
|
||||
for area in sorted(p for p in (kb_dir / collection).iterdir() if p.is_dir()):
|
||||
area.rmdir()
|
||||
|
||||
n = 0
|
||||
for value, count in distribution.items():
|
||||
for _ in range(count):
|
||||
n += 1
|
||||
write_page(
|
||||
kb_dir / collection / f"page-{n:03d}.md",
|
||||
{
|
||||
"type": type_path, field: value,
|
||||
"created": "2026-08-01", "modified": "2026-08-01",
|
||||
"provenance": "general", "summary": f"Page {n}",
|
||||
},
|
||||
f"\n# page-{n:03d}\n",
|
||||
)
|
||||
|
||||
|
||||
def test_lint_recommends_areas_for_a_collection_past_the_threshold(kb_dir):
|
||||
"""The concepts shape: 80 pages in one table, no areas, and a subtype axis
|
||||
whose largest value (28) lands well under the threshold. Sharding is
|
||||
per-area and automatic, so a collection with no areas never splits however
|
||||
large it grows - the threshold is a dead value until someone makes areas."""
|
||||
_fill_collection(
|
||||
kb_dir, "concepts", "types/concept.md", "concept_type",
|
||||
{"workflow": 28, "architecture": 20, "pattern": 17,
|
||||
"decision": 7, "problem": 5, "protocol": 3},
|
||||
)
|
||||
finding = next(
|
||||
i for i in run_lint(kb_dir)["unsharded_collections"] if i["collection"] == "concepts"
|
||||
)
|
||||
assert finding["field"] == "concept_type"
|
||||
# Largest first, so the reader sees the area that decides whether it helps.
|
||||
assert finding["distribution"][0] == {"value": "workflow", "count": 28}
|
||||
assert {d["value"] for d in finding["distribution"]} == {
|
||||
"workflow", "architecture", "pattern", "decision", "problem", "protocol"
|
||||
}
|
||||
assert finding["count"] == sum(d["count"] for d in finding["distribution"])
|
||||
# `types/concept.md` carries the layout, so only the move is still owed -
|
||||
# the report line says which half of the fix that is.
|
||||
assert finding["layout_declared"] is True
|
||||
|
||||
|
||||
def test_the_area_recommendation_says_the_layout_is_missing_when_it_is(kb_dir):
|
||||
"""`types/source.md` deliberately declares no `layout:`, so a sources
|
||||
collection that grew past the threshold has nothing for `move` to compute a
|
||||
destination from - the fix starts one step earlier, and the report says so."""
|
||||
_fill_collection(
|
||||
kb_dir, "sources", "types/source.md", "source_type",
|
||||
{"notes": 30, "article": 21},
|
||||
)
|
||||
report = run_lint(kb_dir)
|
||||
finding = next(i for i in report["unsharded_collections"] if i["collection"] == "sources")
|
||||
assert finding["layout_declared"] is False
|
||||
assert "Declare a `layout:`" in render_markdown(report)
|
||||
|
||||
|
||||
def test_the_area_recommendation_is_not_a_failure(kb_dir):
|
||||
"""It reports a collection that has outgrown a layout, not a broken one.
|
||||
A corpus whose only finding is this must stay green, or every instance
|
||||
goes red on the release that shipped the check."""
|
||||
_fill_collection(
|
||||
kb_dir, "concepts", "types/concept.md", "concept_type",
|
||||
{"workflow": 28, "architecture": 24},
|
||||
)
|
||||
report = run_lint(kb_dir)
|
||||
assert report["unsharded_collections"]
|
||||
assert "unsharded_collections" not in HARD_ERROR_KEYS
|
||||
assert not has_hard_errors({**{key: [] for key in HARD_ERROR_KEYS},
|
||||
"unsharded_collections": report["unsharded_collections"]})
|
||||
|
||||
|
||||
def test_lint_is_silent_about_a_collection_under_the_threshold(kb_dir):
|
||||
"""The sources shape: 29 pages, lopsided across `source_type` - and under
|
||||
the threshold anyway, so it never fires. That is what keeps the bad split
|
||||
(one area of 25 plus four splinters) from ever being recommended, without
|
||||
the check needing to know anything about sources."""
|
||||
_fill_collection(
|
||||
kb_dir, "sources", "types/source.md", "source_type",
|
||||
{"notes": 25, "article": 3, "document": 1},
|
||||
)
|
||||
assert run_lint(kb_dir)["unsharded_collections"] == []
|
||||
|
||||
|
||||
def test_lint_is_silent_when_the_split_would_not_help(kb_dir):
|
||||
"""Past the threshold, but 70 of 80 share one subtype value: splitting
|
||||
produces one area still over the threshold plus splinters, which is not an
|
||||
improvement. The second half of the criterion, and the one a
|
||||
threshold-only check would have got wrong."""
|
||||
_fill_collection(
|
||||
kb_dir, "concepts", "types/concept.md", "concept_type",
|
||||
{"workflow": 70, "architecture": 6, "pattern": 5},
|
||||
)
|
||||
assert run_lint(kb_dir)["unsharded_collections"] == []
|
||||
|
||||
|
||||
def test_lint_is_silent_once_the_collection_has_areas(kb_dir):
|
||||
"""After the fix - the pages sit in their areas - the finding goes away,
|
||||
and `index rebuild` shards whatever outgrows the threshold from here."""
|
||||
_fill_collection(
|
||||
kb_dir, "concepts", "types/concept.md", "concept_type",
|
||||
{"workflow": 28, "architecture": 24},
|
||||
)
|
||||
for page in sorted((kb_dir / "concepts").glob("page-*.md")):
|
||||
area = "workflows" if "workflow" in page.read_text(encoding="utf-8") else "architectures"
|
||||
(kb_dir / "concepts" / area).mkdir(exist_ok=True)
|
||||
page.rename(kb_dir / "concepts" / area / page.name)
|
||||
assert run_lint(kb_dir)["unsharded_collections"] == []
|
||||
|
||||
|
||||
def test_lint_flags_legacy_citation_marker_as_hard_error(kb_dir):
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{"type": "types/concept.md", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": ["Source - Aurora"], "confidence": 0.7},
|
||||
"\n# Modbus\n\n## Definition\n\nUses port 502 ^[[Source - Aurora]].\n",
|
||||
@@ -250,7 +374,7 @@ def test_lint_flags_legacy_citation_marker_as_hard_error(kb_dir):
|
||||
|
||||
def test_lint_flags_undefined_footnote_ref_as_hard_error(kb_dir):
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{"type": "types/concept.md", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": [], "confidence": 0.7},
|
||||
"\n# Modbus\n\n## Definition\n\nUses port 502 [^s-ghost].\n",
|
||||
@@ -264,7 +388,7 @@ def test_lint_flags_orphan_footnote_def_as_hard_error(kb_dir):
|
||||
cid = cite_id("Source - Aurora")
|
||||
block = render_cite_block({cid: ("Source - Aurora", None)})
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{"type": "types/concept.md", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": ["Source - Aurora"], "confidence": 0.7},
|
||||
f"\n# Modbus\n\n## Definition\n\nIndustrial protocol, no citation here.\n\n{block}",
|
||||
@@ -278,7 +402,7 @@ def test_lint_clean_footnote_citation_has_no_hard_errors(kb_dir):
|
||||
cid = cite_id("Source - Aurora")
|
||||
block = render_cite_block({cid: ("Source - Aurora", None)})
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{"type": "types/concept.md", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": ["Source - Aurora"], "confidence": 0.7},
|
||||
f"\n# Modbus\n\n## Definition\n\nUses port 502 [^{cid}].\n\n{block}",
|
||||
|
||||
@@ -43,8 +43,8 @@ def test_page_subdir_falls_back_for_unmapped_subtype():
|
||||
|
||||
|
||||
def test_page_subdir_is_none_for_types_without_layout():
|
||||
assert _page_subdir(None, "types/concept.md") is None
|
||||
assert _page_subdir("anything", "types/concept.md") is None
|
||||
assert _page_subdir(None, "types/source.md") is None
|
||||
assert _page_subdir("anything", "types/source.md") is None
|
||||
|
||||
|
||||
def test_coerce_set_value_uses_declared_schema_type():
|
||||
@@ -258,12 +258,17 @@ def test_new_source_rejects_invalid_source_type(monkeypatch, kb_dir):
|
||||
assert result.exit_code != 0
|
||||
|
||||
|
||||
def test_new_concept_creates_page(monkeypatch, kb_dir):
|
||||
def test_new_concept_creates_page_in_its_subtype_area(monkeypatch, kb_dir):
|
||||
"""Gitea #59: `types/concept.md` declares a `layout:` now, so a new concept
|
||||
reaches its area with nothing else asked of the author - the same rule that
|
||||
has always placed an entity. Nothing about `new` changed to make this true;
|
||||
the type-spec did."""
|
||||
result = _invoke_new(monkeypatch, kb_dir, [
|
||||
"new", "concept", "--name", "Event Sourcing", "--set", "concept_type=pattern",
|
||||
])
|
||||
assert result.exit_code == 0, result.output
|
||||
assert (kb_dir / "concepts/Event Sourcing.md").exists()
|
||||
assert (kb_dir / "concepts/patterns/Event Sourcing.md").exists()
|
||||
assert not (kb_dir / "concepts/Event Sourcing.md").exists()
|
||||
|
||||
|
||||
def test_new_concept_rejects_invalid_concept_type(monkeypatch, kb_dir):
|
||||
|
||||
@@ -310,7 +310,7 @@ def test_citing_pages_via_frontmatter_and_inline(kb_dir, raw_dir):
|
||||
)
|
||||
refs, block = _footnote_block(("Source - Aurora", None))
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{
|
||||
"type": "concept", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": [], "confidence": 0.7,
|
||||
@@ -325,7 +325,7 @@ def test_citing_pages_via_frontmatter_and_inline(kb_dir, raw_dir):
|
||||
|
||||
def test_page_raw_files_resolves_through_sources_and_inline(kb_dir, raw_dir):
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{
|
||||
"type": "concept", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": ["Source - Aurora"], "confidence": 0.7,
|
||||
@@ -429,7 +429,7 @@ def test_lint_flags_citation_not_in_frontmatter_sources(kb_dir, raw_dir, monkeyp
|
||||
|
||||
refs, block = _footnote_block(("Source - Aurora", None))
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{
|
||||
"type": "concept", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": [], "confidence": 0.7,
|
||||
@@ -451,7 +451,7 @@ def test_lint_no_drift_when_source_declared(kb_dir, raw_dir, monkeypatch):
|
||||
|
||||
refs, block = _footnote_block(("Source - Aurora", None))
|
||||
write_page(
|
||||
kb_dir / "concepts/Modbus.md",
|
||||
kb_dir / "concepts/protocols/Modbus.md",
|
||||
{
|
||||
"type": "concept", "concept_type": "protocol", "tags": [], "created": "2026-07-25",
|
||||
"modified": "2026-07-25", "related": [], "sources": ["Source - Aurora"], "confidence": 0.7,
|
||||
|
||||
@@ -86,9 +86,34 @@ def test_get_layout_reads_entity_type_specs_own_layout_field():
|
||||
assert list(layout) == ["project", "system", "tool", "technology", "person"]
|
||||
|
||||
|
||||
def test_concept_layout_covers_every_declared_concept_type():
|
||||
"""Gitea #59: `kb/concepts/` had no areas, so the catalog's per-area shard
|
||||
threshold could never fire however large it grew. A value missing from the
|
||||
layout would still be *placed* (`subtype_dir` pluralizes the fallback), but
|
||||
into an area with no declared title - so the schema's enum and the layout
|
||||
have to agree, and this is what checks that they do."""
|
||||
layout = resolver.get_layout("types/concept.md")
|
||||
assert layout is not None
|
||||
assert set(layout) == set(resolver.get_enum("types/concept.md", "concept_type"))
|
||||
# `dir` is structural; `title` is display text following the KB language, so
|
||||
# it is checked for presence rather than wording (see the entity test above).
|
||||
assert {key: spec["dir"] for key, spec in layout.items()} == {
|
||||
"architecture": "architectures",
|
||||
"pattern": "patterns",
|
||||
"protocol": "protocols",
|
||||
"workflow": "workflows",
|
||||
"decision": "decisions",
|
||||
"problem": "problems",
|
||||
}
|
||||
assert all(spec.get("title") for spec in layout.values())
|
||||
|
||||
|
||||
def test_get_layout_is_none_for_types_without_one():
|
||||
"""`comparison` has no subtype field at all; `source` has one and
|
||||
deliberately declares no `layout:` anyway - 25 of its 29 pages carry the
|
||||
same `source_type`, so splitting on it would make one area and four
|
||||
splinters (Gitea #59). Having a subtype axis is not a reason to use it."""
|
||||
assert resolver.get_layout("types/comparison.md") is None
|
||||
assert resolver.get_layout("types/concept.md") is None
|
||||
assert resolver.get_layout("types/source.md") is None
|
||||
|
||||
|
||||
@@ -237,7 +262,7 @@ def test_subtype_dir_falls_back_for_unmapped_subtype():
|
||||
|
||||
|
||||
def test_subtype_dir_is_none_without_layout_or_subtype():
|
||||
assert resolver.subtype_dir("types/concept.md", "workflow") is None
|
||||
assert resolver.subtype_dir("types/source.md", "notes") is None
|
||||
assert resolver.subtype_dir("types/entity.md", None) is None
|
||||
|
||||
|
||||
@@ -247,8 +272,8 @@ def test_compute_target_dir_applies_layout_subdirectory():
|
||||
|
||||
|
||||
def test_compute_target_dir_is_flat_for_a_type_without_layout():
|
||||
target = resolver.compute_target_dir("types/concept.md", {"concept_type": "workflow"})
|
||||
assert target == config.KB_DIR / "concepts"
|
||||
target = resolver.compute_target_dir("types/source.md", {"source_type": "notes"})
|
||||
assert target == config.KB_DIR / "sources"
|
||||
|
||||
|
||||
def test_compute_target_dir_resolves_against_repo_root_for_root_repo_types():
|
||||
|
||||
@@ -121,7 +121,7 @@ def test_xref_add_updates_both_pages_on_disk(kb_dir):
|
||||
config.INDEX_FILE = kb_dir / "index.md"
|
||||
|
||||
runner = CliRunner()
|
||||
modbus_before = (kb_dir / "concepts/Modbus.md").read_text(encoding="utf-8")
|
||||
modbus_before = (kb_dir / "concepts/protocols/Modbus.md").read_text(encoding="utf-8")
|
||||
result = runner.invoke(app, ["xref", "add", "--a", "gdeploy", "--b", "Modbus", "--rel", "uses"])
|
||||
assert result.exit_code == 0, result.output
|
||||
|
||||
@@ -131,7 +131,7 @@ def test_xref_add_updates_both_pages_on_disk(kb_dir):
|
||||
|
||||
# B is not touched at all. Its inbound view is rendered from the graph, so
|
||||
# nothing has to be written there for a reader to find its way back.
|
||||
assert (kb_dir / "concepts/Modbus.md").read_text(encoding="utf-8") == modbus_before
|
||||
assert (kb_dir / "concepts/protocols/Modbus.md").read_text(encoding="utf-8") == modbus_before
|
||||
|
||||
_fm_before, body_before = read_page(kb_dir / "entities/tools/gdeploy.md")
|
||||
link_count_before = body_before.count("[[Modbus]]")
|
||||
@@ -238,7 +238,7 @@ def test_xref_add_dry_run_writes_nothing(kb_dir):
|
||||
config.INDEX_FILE = kb_dir / "index.md"
|
||||
|
||||
gdeploy_path = kb_dir / "entities/tools/gdeploy.md"
|
||||
modbus_path = kb_dir / "concepts/Modbus.md"
|
||||
modbus_path = kb_dir / "concepts/protocols/Modbus.md"
|
||||
gdeploy_before = gdeploy_path.read_text(encoding="utf-8")
|
||||
modbus_before = modbus_path.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user