Files
chemenu/tools/chemenu/catalog.py
T
torben 7f74303a00
CI / verify (push) Successful in 52s
Release / release (push) Successful in 36s
kb/concepts/ bekommt Areas: layout: fuer concept, Area-Titel aus jedem Type-Spec, Schwellen-Empfehlung im lint (schliesst #59)
Files changed:
- CHANGES.md
- README.md
- VERSION
- kb/concepts/Ambient Environment Dependency.md
- kb/concepts/Anti-Cramming Heuristic.md
- kb/concepts/Audit Trail.md
- kb/concepts/BM25.md
- kb/concepts/Bulk Operations.md
- kb/concepts/CI Integration.md
- kb/concepts/COLLECTION.md
- kb/concepts/CPPC.md
- kb/concepts/Checkpoint Audit.md
- kb/concepts/Claude Code Auto Mode.md
- kb/concepts/Command Round-Trip Integrity.md
- kb/concepts/Confidence Scoring.md
- kb/concepts/Consolidation Tiers.md
- kb/concepts/Content Quality Control.md
- kb/concepts/Context Isolation.md
- kb/concepts/Contradiction Resolution.md
- kb/concepts/Cross-platform Agent Skills.md
- kb/concepts/Crystallization.md
- kb/concepts/Delete Rather Than Anonymize.md
- kb/concepts/Denylist over Allowlist.md
- kb/concepts/Detect-Repair Asymmetry.md
- kb/concepts/Diff-Reviewable Agent Edits.md
- kb/concepts/Dual Licensing by File Plan.md
- kb/concepts/Entity Extraction.md
- kb/concepts/Episodic Memory.md
- kb/concepts/Event-Driven Automation.md
- kb/concepts/Filter on Ingest.md
- kb/concepts/Forgetting.md
- kb/concepts/Graph Traversal.md
- kb/concepts/Green Suite Blind Spot.md
- kb/concepts/Hooks.md
- kb/concepts/Hybrid Search.md
- kb/concepts/INDEX.md
- kb/concepts/Implementation Spectrum.md
- kb/concepts/Index Scaling.md
- kb/concepts/Issue Label Scheme.md
- kb/concepts/Iteration and Cost Limits.md
- kb/concepts/KB Migration.md
- kb/concepts/KB Stack Versioning.md
- kb/concepts/Knowledge Compounding.md
- kb/concepts/Knowledge Graph.md
- kb/concepts/LLM Wiki Pattern.md
- kb/concepts/Lint Workflow.md
- kb/concepts/MCP-Leseserver.md
- kb/concepts/Mass-Update Gate.md
- kb/concepts/Memory Lifecycle.md
- kb/concepts/Mesh Sync.md
- kb/concepts/Modbus.md
- kb/concepts/Multi-Agent Collaboration.md
- kb/concepts/Naming Convention Conflict.md
- kb/concepts/OKF Compatibility.md
- kb/concepts/Optional Instance Context File.md
- kb/concepts/Personalization Plane.md
- kb/concepts/Privacy and Governance.md
- kb/concepts/Procedural Memory.md
- kb/concepts/Publish-Remote Gate.md
- kb/concepts/Quality Scoring.md
- kb/concepts/Quality and Self-Correction.md
- kb/concepts/RAG.md
- kb/concepts/Reciprocal Rank Fusion.md
- kb/concepts/SSD TRIM.md
- kb/concepts/Scale Ceiling.md
- kb/concepts/Self-Healing.md
- kb/concepts/Semantic Lint Automation.md
- kb/concepts/Semantic Memory.md
- kb/concepts/Session Orientation.md
- kb/concepts/Shared vs Private.md
- kb/concepts/Split Merge Reclassify.md
- kb/concepts/Split Threshold.md
- kb/concepts/Structural Enforcement over Documented Rule.md
- kb/concepts/Stub Threshold.md
- kb/concepts/Supersession.md
- kb/concepts/Three-Layer Architecture.md
- kb/concepts/Token Economics.md
- kb/concepts/Typed Relationships.md
- kb/concepts/User Management.md
- kb/concepts/Vector Search.md
- kb/concepts/Work Coordination.md
- kb/concepts/Workflow Extraction.md
- kb/concepts/Workflow Orchestration.md
- kb/concepts/Working Memory.md
- kb/concepts/Write-Once Frontmatter Fields.md
- kb/concepts/architectures/Consolidation Tiers.md
- kb/concepts/architectures/Context Isolation.md
- kb/concepts/architectures/Cross-platform Agent Skills.md
- kb/concepts/architectures/Episodic Memory.md
- kb/concepts/architectures/Hybrid Search.md
- kb/concepts/architectures/Implementation Spectrum.md
- kb/concepts/architectures/Knowledge Graph.md
- kb/concepts/architectures/LLM Wiki Pattern.md
- kb/concepts/architectures/MCP-Leseserver.md
- kb/concepts/architectures/Memory Lifecycle.md
- kb/concepts/architectures/OKF Compatibility.md
- kb/concepts/architectures/Optional Instance Context File.md
- kb/concepts/architectures/Personalization Plane.md
- kb/concepts/architectures/Procedural Memory.md
- kb/concepts/architectures/RAG.md
- kb/concepts/architectures/Scale Ceiling.md
- kb/concepts/architectures/Semantic Memory.md
- kb/concepts/architectures/Three-Layer Architecture.md
- kb/concepts/architectures/Token Economics.md
- kb/concepts/architectures/Working Memory.md
- kb/concepts/decisions/Delete Rather Than Anonymize.md
- kb/concepts/decisions/Denylist over Allowlist.md
- kb/concepts/decisions/Diff-Reviewable Agent Edits.md
- kb/concepts/decisions/Dual Licensing by File Plan.md
- kb/concepts/decisions/Issue Label Scheme.md
- kb/concepts/decisions/KB Stack Versioning.md
- kb/concepts/decisions/Structural Enforcement over Documented Rule.md
- kb/concepts/patterns/Audit Trail.md
- kb/concepts/patterns/BM25.md
- kb/concepts/patterns/Command Round-Trip Integrity.md
- kb/concepts/patterns/Confidence Scoring.md
- kb/concepts/patterns/Contradiction Resolution.md
- kb/concepts/patterns/Entity Extraction.md
- kb/concepts/patterns/Filter on Ingest.md
- kb/concepts/patterns/Forgetting.md
- kb/concepts/patterns/Graph Traversal.md
- kb/concepts/patterns/Mesh Sync.md
- kb/concepts/patterns/Quality Scoring.md
- kb/concepts/patterns/Reciprocal Rank Fusion.md
- kb/concepts/patterns/Self-Healing.md
- kb/concepts/patterns/Shared vs Private.md
- kb/concepts/patterns/Typed Relationships.md
- kb/concepts/patterns/Vector Search.md
- kb/concepts/patterns/Work Coordination.md
- kb/concepts/problems/Ambient Environment Dependency.md
- kb/concepts/problems/Detect-Repair Asymmetry.md
- kb/concepts/problems/Green Suite Blind Spot.md
- kb/concepts/problems/Naming Convention Conflict.md
- kb/concepts/problems/Write-Once Frontmatter Fields.md
- kb/concepts/protocols/CPPC.md
- kb/concepts/protocols/Modbus.md
- kb/concepts/protocols/SSD TRIM.md
- kb/concepts/workflows/Anti-Cramming Heuristic.md
- kb/concepts/workflows/Bulk Operations.md
- kb/concepts/workflows/CI Integration.md
- kb/concepts/workflows/Checkpoint Audit.md
- kb/concepts/workflows/Claude Code Auto Mode.md
- kb/concepts/workflows/Content Quality Control.md
- kb/concepts/workflows/Crystallization.md
- kb/concepts/workflows/Event-Driven Automation.md
- kb/concepts/workflows/Hooks.md
- kb/concepts/workflows/Index Scaling.md
- kb/concepts/workflows/Iteration and Cost Limits.md
- kb/concepts/workflows/KB Migration.md
- kb/concepts/workflows/Knowledge Compounding.md
- kb/concepts/workflows/Lint Workflow.md
- kb/concepts/workflows/Mass-Update Gate.md
- kb/concepts/workflows/Multi-Agent Collaboration.md
- kb/concepts/workflows/Privacy and Governance.md
- kb/concepts/workflows/Publish-Remote Gate.md
- kb/concepts/workflows/Quality and Self-Correction.md
- kb/concepts/workflows/Semantic Lint Automation.md
- kb/concepts/workflows/Session Orientation.md
- kb/concepts/workflows/Split Merge Reclassify.md
- kb/concepts/workflows/Split Threshold.md
- kb/concepts/workflows/Stub Threshold.md
- kb/concepts/workflows/Supersession.md
- kb/concepts/workflows/User Management.md
- kb/concepts/workflows/Workflow Extraction.md
- kb/concepts/workflows/Workflow Orchestration.md
- kb/index.md
- kb/log.md
- tools/CONTRACT.md
- tools/README.md
- tools/chemenu/catalog.py
- tools/chemenu/commands/index_build.py
- tools/chemenu/lint_core.py
- tools/chemenu/tests/conftest.py
- tools/chemenu/tests/test_cite_cmd.py
- tools/chemenu/tests/test_git_publish.py
- tools/chemenu/tests/test_index_build.py
- tools/chemenu/tests/test_lint.py
- tools/chemenu/tests/test_new_page.py
- tools/chemenu/tests/test_provenance.py
- tools/chemenu/tests/test_type_resolver.py
- tools/chemenu/tests/test_xref.py
- types/concept.md
- types/type-spec.md
2026-09-08 10:07:46 +02:00

141 lines
5.5 KiB
Python

"""How the corpus groups into collections and areas, with no CLI attached.
Split out of `commands/index_build.py` for the reason `lint_core.py` gives at
the top of itself: this is a pure function over a corpus directory, and it was
sitting in a module that imports `typer` and `rich`. `lint` needs the same
grouping - it is what answers "does this collection have areas, and is it over
the threshold?" (Gitea #59) - and `chemenu.api`, the read surface, may not
reach a command module at all. Importing it from there would have pulled the
whole CLI head in behind it.
So the split runs along the same line as lint's: everything that decides *how
the corpus is shaped* lives here; everything that decides *what the catalog
looks like* - the tables, the map, the shard files - stays in
`commands/index_build.py`, which imports from here.
"""
from __future__ import annotations
from dataclasses import dataclass, field
from pathlib import Path
from chemenu.kb_collections import iter_kb_collections
from chemenu.page import Page
from chemenu.type_resolver import resolver
# Rows per area before it is split into its own shard. From the wiki's own
# `Index Scaling` page ("split table sections at >50 entries"), kept as a plain
# number so growth is handled by arithmetic rather than by a judgment call.
SHARD_THRESHOLD = 50
# Display title for pages sitting directly in a collection root rather than in
# an area subdirectory.
UNGROUPED_TITLE = "All"
@dataclass
class Area:
"""One grouping inside a collection: a subdirectory, or the collection root
for pages that sit directly in it."""
name: str
title: str
pages: list[Page] = field(default_factory=list)
own_shard: bool = False
@property
def count(self) -> int:
return len(self.pages)
@dataclass
class Collection:
name: str
areas: list[Area] = field(default_factory=list)
@property
def count(self) -> int:
return sum(area.count for area in self.areas)
def area_titles() -> dict[str, dict[str, str]]:
"""Display titles per collection: `{collection: {area_dir: title}}`, taken
from each type-spec's own `layout:` rather than a hardcoded map - so a new
subtype names its own section by adding a type-spec, with no code change.
Every type-spec is read, not just `entity`'s. That hardcoding was the
asymmetry behind Gitea #59: the axis a collection splits along is declared
in `layout:`, and a second type declaring one would have had its areas
titled by `.title()` on the directory name while entity's got their real
names.
Keyed by collection rather than by directory name alone, because two types
writing into two collections may legitimately use the same area name for
different things (`kb/entities/tools/` and a hypothetical
`kb/concepts/tools/`); a flat map would hand the second one the first's
title. The collection key is the type's `base_dir:`, which is what put the
page in that directory to begin with.
"""
titles: dict[str, dict[str, str]] = {}
for type_path, _frontmatter in resolver.list_type_specs():
try:
if resolver.get_root(type_path) != "kb":
continue
base_dir = resolver.get_base_dir(type_path)
layout = resolver.get_layout(type_path)
except (ValueError, OSError):
continue
if not base_dir or not layout:
continue
per_collection = titles.setdefault(str(base_dir).strip("/"), {})
for key, spec in layout.items():
per_collection.setdefault(spec.get("dir", key), spec.get("title", str(key).title()))
return titles
def group_pages(kb_dir: Path, pages: dict[str, Page]) -> list[Collection]:
"""Group pages by their physical location: collection directory, then area
subdirectory.
Location rather than `kind` because a shard lives in the directory it
describes, and the two agree by construction: a type-spec's `base_dir:` is
what put the page there.
"""
titles = area_titles()
grouped: dict[str, dict[str, Area]] = {}
# Seed from the collections that exist on disk, not only from the ones that
# happen to hold pages: an empty collection is a real (if unfilled) part of
# the wiki, and dropping it from the map would hide it from every reader.
for collection_dir in iter_kb_collections(kb_dir):
grouped.setdefault(collection_dir.name, {})
for page in sorted(pages.values(), key=lambda p: p.title.lower()):
try:
parts = page.path.relative_to(kb_dir).parts
except ValueError: # pragma: no cover - pages always live under kb_dir
continue
if len(parts) < 2:
collection_name, area_name = "(kb root)", ""
else:
collection_name = parts[0]
area_name = parts[1] if len(parts) > 2 else ""
areas = grouped.setdefault(collection_name, {})
area = areas.get(area_name)
if area is None:
title = (
titles.get(collection_name, {}).get(area_name, area_name.title())
if area_name
else UNGROUPED_TITLE
)
area = Area(name=area_name, title=title)
areas[area_name] = area
area.pages.append(page)
collections = []
for name in sorted(grouped):
ordered = sorted(grouped[name].values(), key=lambda a: (a.name == "", a.title.lower()))
for area in ordered:
area.own_shard = bool(area.name) and area.count > SHARD_THRESHOLD
collections.append(Collection(name=name, areas=ordered))
return collections