Files changed: - CHANGES.md - README.md - VERSION - kb/concepts/Ambient Environment Dependency.md - kb/concepts/Anti-Cramming Heuristic.md - kb/concepts/Audit Trail.md - kb/concepts/BM25.md - kb/concepts/Bulk Operations.md - kb/concepts/CI Integration.md - kb/concepts/COLLECTION.md - kb/concepts/CPPC.md - kb/concepts/Checkpoint Audit.md - kb/concepts/Claude Code Auto Mode.md - kb/concepts/Command Round-Trip Integrity.md - kb/concepts/Confidence Scoring.md - kb/concepts/Consolidation Tiers.md - kb/concepts/Content Quality Control.md - kb/concepts/Context Isolation.md - kb/concepts/Contradiction Resolution.md - kb/concepts/Cross-platform Agent Skills.md - kb/concepts/Crystallization.md - kb/concepts/Delete Rather Than Anonymize.md - kb/concepts/Denylist over Allowlist.md - kb/concepts/Detect-Repair Asymmetry.md - kb/concepts/Diff-Reviewable Agent Edits.md - kb/concepts/Dual Licensing by File Plan.md - kb/concepts/Entity Extraction.md - kb/concepts/Episodic Memory.md - kb/concepts/Event-Driven Automation.md - kb/concepts/Filter on Ingest.md - kb/concepts/Forgetting.md - kb/concepts/Graph Traversal.md - kb/concepts/Green Suite Blind Spot.md - kb/concepts/Hooks.md - kb/concepts/Hybrid Search.md - kb/concepts/INDEX.md - kb/concepts/Implementation Spectrum.md - kb/concepts/Index Scaling.md - kb/concepts/Issue Label Scheme.md - kb/concepts/Iteration and Cost Limits.md - kb/concepts/KB Migration.md - kb/concepts/KB Stack Versioning.md - kb/concepts/Knowledge Compounding.md - kb/concepts/Knowledge Graph.md - kb/concepts/LLM Wiki Pattern.md - kb/concepts/Lint Workflow.md - kb/concepts/MCP-Leseserver.md - kb/concepts/Mass-Update Gate.md - kb/concepts/Memory Lifecycle.md - kb/concepts/Mesh Sync.md - kb/concepts/Modbus.md - kb/concepts/Multi-Agent Collaboration.md - kb/concepts/Naming Convention Conflict.md - kb/concepts/OKF Compatibility.md - kb/concepts/Optional Instance Context File.md - kb/concepts/Personalization Plane.md - kb/concepts/Privacy and Governance.md - kb/concepts/Procedural Memory.md - kb/concepts/Publish-Remote Gate.md - kb/concepts/Quality Scoring.md - kb/concepts/Quality and Self-Correction.md - kb/concepts/RAG.md - kb/concepts/Reciprocal Rank Fusion.md - kb/concepts/SSD TRIM.md - kb/concepts/Scale Ceiling.md - kb/concepts/Self-Healing.md - kb/concepts/Semantic Lint Automation.md - kb/concepts/Semantic Memory.md - kb/concepts/Session Orientation.md - kb/concepts/Shared vs Private.md - kb/concepts/Split Merge Reclassify.md - kb/concepts/Split Threshold.md - kb/concepts/Structural Enforcement over Documented Rule.md - kb/concepts/Stub Threshold.md - kb/concepts/Supersession.md - kb/concepts/Three-Layer Architecture.md - kb/concepts/Token Economics.md - kb/concepts/Typed Relationships.md - kb/concepts/User Management.md - kb/concepts/Vector Search.md - kb/concepts/Work Coordination.md - kb/concepts/Workflow Extraction.md - kb/concepts/Workflow Orchestration.md - kb/concepts/Working Memory.md - kb/concepts/Write-Once Frontmatter Fields.md - kb/concepts/architectures/Consolidation Tiers.md - kb/concepts/architectures/Context Isolation.md - kb/concepts/architectures/Cross-platform Agent Skills.md - kb/concepts/architectures/Episodic Memory.md - kb/concepts/architectures/Hybrid Search.md - kb/concepts/architectures/Implementation Spectrum.md - kb/concepts/architectures/Knowledge Graph.md - kb/concepts/architectures/LLM Wiki Pattern.md - kb/concepts/architectures/MCP-Leseserver.md - kb/concepts/architectures/Memory Lifecycle.md - kb/concepts/architectures/OKF Compatibility.md - kb/concepts/architectures/Optional Instance Context File.md - kb/concepts/architectures/Personalization Plane.md - kb/concepts/architectures/Procedural Memory.md - kb/concepts/architectures/RAG.md - kb/concepts/architectures/Scale Ceiling.md - kb/concepts/architectures/Semantic Memory.md - kb/concepts/architectures/Three-Layer Architecture.md - kb/concepts/architectures/Token Economics.md - kb/concepts/architectures/Working Memory.md - kb/concepts/decisions/Delete Rather Than Anonymize.md - kb/concepts/decisions/Denylist over Allowlist.md - kb/concepts/decisions/Diff-Reviewable Agent Edits.md - kb/concepts/decisions/Dual Licensing by File Plan.md - kb/concepts/decisions/Issue Label Scheme.md - kb/concepts/decisions/KB Stack Versioning.md - kb/concepts/decisions/Structural Enforcement over Documented Rule.md - kb/concepts/patterns/Audit Trail.md - kb/concepts/patterns/BM25.md - kb/concepts/patterns/Command Round-Trip Integrity.md - kb/concepts/patterns/Confidence Scoring.md - kb/concepts/patterns/Contradiction Resolution.md - kb/concepts/patterns/Entity Extraction.md - kb/concepts/patterns/Filter on Ingest.md - kb/concepts/patterns/Forgetting.md - kb/concepts/patterns/Graph Traversal.md - kb/concepts/patterns/Mesh Sync.md - kb/concepts/patterns/Quality Scoring.md - kb/concepts/patterns/Reciprocal Rank Fusion.md - kb/concepts/patterns/Self-Healing.md - kb/concepts/patterns/Shared vs Private.md - kb/concepts/patterns/Typed Relationships.md - kb/concepts/patterns/Vector Search.md - kb/concepts/patterns/Work Coordination.md - kb/concepts/problems/Ambient Environment Dependency.md - kb/concepts/problems/Detect-Repair Asymmetry.md - kb/concepts/problems/Green Suite Blind Spot.md - kb/concepts/problems/Naming Convention Conflict.md - kb/concepts/problems/Write-Once Frontmatter Fields.md - kb/concepts/protocols/CPPC.md - kb/concepts/protocols/Modbus.md - kb/concepts/protocols/SSD TRIM.md - kb/concepts/workflows/Anti-Cramming Heuristic.md - kb/concepts/workflows/Bulk Operations.md - kb/concepts/workflows/CI Integration.md - kb/concepts/workflows/Checkpoint Audit.md - kb/concepts/workflows/Claude Code Auto Mode.md - kb/concepts/workflows/Content Quality Control.md - kb/concepts/workflows/Crystallization.md - kb/concepts/workflows/Event-Driven Automation.md - kb/concepts/workflows/Hooks.md - kb/concepts/workflows/Index Scaling.md - kb/concepts/workflows/Iteration and Cost Limits.md - kb/concepts/workflows/KB Migration.md - kb/concepts/workflows/Knowledge Compounding.md - kb/concepts/workflows/Lint Workflow.md - kb/concepts/workflows/Mass-Update Gate.md - kb/concepts/workflows/Multi-Agent Collaboration.md - kb/concepts/workflows/Privacy and Governance.md - kb/concepts/workflows/Publish-Remote Gate.md - kb/concepts/workflows/Quality and Self-Correction.md - kb/concepts/workflows/Semantic Lint Automation.md - kb/concepts/workflows/Session Orientation.md - kb/concepts/workflows/Split Merge Reclassify.md - kb/concepts/workflows/Split Threshold.md - kb/concepts/workflows/Stub Threshold.md - kb/concepts/workflows/Supersession.md - kb/concepts/workflows/User Management.md - kb/concepts/workflows/Workflow Extraction.md - kb/concepts/workflows/Workflow Orchestration.md - kb/index.md - kb/log.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/catalog.py - tools/chemenu/commands/index_build.py - tools/chemenu/lint_core.py - tools/chemenu/tests/conftest.py - tools/chemenu/tests/test_cite_cmd.py - tools/chemenu/tests/test_git_publish.py - tools/chemenu/tests/test_index_build.py - tools/chemenu/tests/test_lint.py - tools/chemenu/tests/test_new_page.py - tools/chemenu/tests/test_provenance.py - tools/chemenu/tests/test_type_resolver.py - tools/chemenu/tests/test_xref.py - types/concept.md - types/type-spec.md
141 lines
5.5 KiB
Python
141 lines
5.5 KiB
Python
"""How the corpus groups into collections and areas, with no CLI attached.
|
|
|
|
Split out of `commands/index_build.py` for the reason `lint_core.py` gives at
|
|
the top of itself: this is a pure function over a corpus directory, and it was
|
|
sitting in a module that imports `typer` and `rich`. `lint` needs the same
|
|
grouping - it is what answers "does this collection have areas, and is it over
|
|
the threshold?" (Gitea #59) - and `chemenu.api`, the read surface, may not
|
|
reach a command module at all. Importing it from there would have pulled the
|
|
whole CLI head in behind it.
|
|
|
|
So the split runs along the same line as lint's: everything that decides *how
|
|
the corpus is shaped* lives here; everything that decides *what the catalog
|
|
looks like* - the tables, the map, the shard files - stays in
|
|
`commands/index_build.py`, which imports from here.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
|
|
from chemenu.kb_collections import iter_kb_collections
|
|
from chemenu.page import Page
|
|
from chemenu.type_resolver import resolver
|
|
|
|
# Rows per area before it is split into its own shard. From the wiki's own
|
|
# `Index Scaling` page ("split table sections at >50 entries"), kept as a plain
|
|
# number so growth is handled by arithmetic rather than by a judgment call.
|
|
SHARD_THRESHOLD = 50
|
|
|
|
# Display title for pages sitting directly in a collection root rather than in
|
|
# an area subdirectory.
|
|
UNGROUPED_TITLE = "All"
|
|
|
|
|
|
@dataclass
|
|
class Area:
|
|
"""One grouping inside a collection: a subdirectory, or the collection root
|
|
for pages that sit directly in it."""
|
|
|
|
name: str
|
|
title: str
|
|
pages: list[Page] = field(default_factory=list)
|
|
own_shard: bool = False
|
|
|
|
@property
|
|
def count(self) -> int:
|
|
return len(self.pages)
|
|
|
|
|
|
@dataclass
|
|
class Collection:
|
|
name: str
|
|
areas: list[Area] = field(default_factory=list)
|
|
|
|
@property
|
|
def count(self) -> int:
|
|
return sum(area.count for area in self.areas)
|
|
|
|
|
|
def area_titles() -> dict[str, dict[str, str]]:
|
|
"""Display titles per collection: `{collection: {area_dir: title}}`, taken
|
|
from each type-spec's own `layout:` rather than a hardcoded map - so a new
|
|
subtype names its own section by adding a type-spec, with no code change.
|
|
|
|
Every type-spec is read, not just `entity`'s. That hardcoding was the
|
|
asymmetry behind Gitea #59: the axis a collection splits along is declared
|
|
in `layout:`, and a second type declaring one would have had its areas
|
|
titled by `.title()` on the directory name while entity's got their real
|
|
names.
|
|
|
|
Keyed by collection rather than by directory name alone, because two types
|
|
writing into two collections may legitimately use the same area name for
|
|
different things (`kb/entities/tools/` and a hypothetical
|
|
`kb/concepts/tools/`); a flat map would hand the second one the first's
|
|
title. The collection key is the type's `base_dir:`, which is what put the
|
|
page in that directory to begin with.
|
|
"""
|
|
titles: dict[str, dict[str, str]] = {}
|
|
for type_path, _frontmatter in resolver.list_type_specs():
|
|
try:
|
|
if resolver.get_root(type_path) != "kb":
|
|
continue
|
|
base_dir = resolver.get_base_dir(type_path)
|
|
layout = resolver.get_layout(type_path)
|
|
except (ValueError, OSError):
|
|
continue
|
|
if not base_dir or not layout:
|
|
continue
|
|
per_collection = titles.setdefault(str(base_dir).strip("/"), {})
|
|
for key, spec in layout.items():
|
|
per_collection.setdefault(spec.get("dir", key), spec.get("title", str(key).title()))
|
|
return titles
|
|
|
|
|
|
def group_pages(kb_dir: Path, pages: dict[str, Page]) -> list[Collection]:
|
|
"""Group pages by their physical location: collection directory, then area
|
|
subdirectory.
|
|
|
|
Location rather than `kind` because a shard lives in the directory it
|
|
describes, and the two agree by construction: a type-spec's `base_dir:` is
|
|
what put the page there.
|
|
"""
|
|
titles = area_titles()
|
|
grouped: dict[str, dict[str, Area]] = {}
|
|
|
|
# Seed from the collections that exist on disk, not only from the ones that
|
|
# happen to hold pages: an empty collection is a real (if unfilled) part of
|
|
# the wiki, and dropping it from the map would hide it from every reader.
|
|
for collection_dir in iter_kb_collections(kb_dir):
|
|
grouped.setdefault(collection_dir.name, {})
|
|
|
|
for page in sorted(pages.values(), key=lambda p: p.title.lower()):
|
|
try:
|
|
parts = page.path.relative_to(kb_dir).parts
|
|
except ValueError: # pragma: no cover - pages always live under kb_dir
|
|
continue
|
|
if len(parts) < 2:
|
|
collection_name, area_name = "(kb root)", ""
|
|
else:
|
|
collection_name = parts[0]
|
|
area_name = parts[1] if len(parts) > 2 else ""
|
|
areas = grouped.setdefault(collection_name, {})
|
|
area = areas.get(area_name)
|
|
if area is None:
|
|
title = (
|
|
titles.get(collection_name, {}).get(area_name, area_name.title())
|
|
if area_name
|
|
else UNGROUPED_TITLE
|
|
)
|
|
area = Area(name=area_name, title=title)
|
|
areas[area_name] = area
|
|
area.pages.append(page)
|
|
|
|
collections = []
|
|
for name in sorted(grouped):
|
|
ordered = sorted(grouped[name].values(), key=lambda a: (a.name == "", a.title.lower()))
|
|
for area in ordered:
|
|
area.own_shard = bool(area.name) and area.count > SHARD_THRESHOLD
|
|
collections.append(Collection(name=name, areas=ordered))
|
|
return collections
|