Chemenu 2.1.0 - deterministischer Wissenskompiler
CI / verify (push) Failing after 32s
Release / release (push) Successful in 38s

Chemenu kompiliert Rohnotizen zu einem verlinkten, quellengebundenen Wiki:
raw/ -> types/ + tools/ -> kb/ -> reports/. Was mechanisch ist, macht
tools/wikitool; was Urteil braucht, macht ein Agent unter Contracts, deren
Grenzen in Code durchgesetzt sind statt im Prompt.

Dieser Commit ist der Startpunkt der oeffentlichen Historie. Die vorherige
Entwicklung fand in einer privaten Instanz statt und ist nicht Teil dieses
Repositorys; ihre Erzaehlung steht vollstaendig in CHANGES.md, das mit 44
Eintraegen von 0.1.0 bis 2.1.0 erhalten geblieben ist.

Der mitgelieferte Korpus ist ein Testbett und eine Demo: 170 Seiten ueber den
Stack selbst - Gates, Lint, Versionierung, Suche, das Wiki-Muster. Er
dokumentiert das Werkzeug mit den eigenen Mitteln des Werkzeugs.

Lizenz: AGPL-3.0 fuer den Stack (tools/, types/), CC-BY-4.0 fuer die Inhalte.
Die Grenze zwischen beiden ist der Dateiplan, den dist export berechnet -
siehe NOTICE.
This commit is contained in:
torben committed 2026-09-01 16:26:14 +02:00
commit 18ae28f918
368 files changed
+50628

No files matched your search

+127
View File
@@ -0,0 +1,127 @@
"""Scoring a session: the structural state it left, and the path it took there.
Two levels, deliberately both hard-oracle:
- **L1, structure.** Counters from `lint`, which answers what the tree looks
like now. It is the same report `lint --fail-on-error` reads, so a score and a
lint run can never disagree.
- **L2, trajectory.** Rules over the trace, which answer how the tree got that
way. This is the half unit tests cannot reach.
There is no judge here and no rubric. Soft-oracle scoring waits until a failure
taxonomy exists - see EVALS.md on the phase gate.
"""
from __future__ import annotations
from datetime import date
from pathlib import Path
from chemenu import config
from chemenu.commands.lint import HARD_ERROR_KEYS, has_hard_errors, run_lint
from chemenu.evals import trajectory
from chemenu.telemetry import reader
# Advisory structural findings: real, but not a broken tree. Kept separate so a
# score reports them without failing on them.
ADVISORY_KEYS = (
"orphan_pages",
"quote_limit_violations",
"uncovered_raw_files",
"unmarked_provenance",
"missing_from_index",
"title_mismatches",
)
def structural_score(kb_dir: Path | None = None, report: dict | None = None) -> dict:
report = report if report is not None else run_lint(kb_dir or config.KB_DIR)
return {
"page_count": report["page_count"],
"hard_errors": has_hard_errors(report),
"errors": {key: len(report.get(key) or []) for key in HARD_ERROR_KEYS},
"advisories": {key: len(report.get(key) or []) for key in ADVISORY_KEYS},
}
def trace_summary(records: list[dict]) -> dict:
counts: dict[str, int] = {}
sources: list[str] = []
for record in records:
event = record.get("event", "?")
counts[event] = counts.get(event, 0) + 1
source = record.get("source")
if source and source not in sources:
sources.append(source)
return {
"events": len(records),
"sources": sources,
"by_event": dict(sorted(counts.items())),
"completeness": reader.completeness(records),
"first": records[0]["ts"] if records else None,
"last": records[-1]["ts"] if records else None,
}
def score(session: str, records: list[dict] | None = None,
kb_dir: Path | None = None, report: dict | None = None) -> dict:
records = reader.read_trace(session) if records is None else records
rules = [rule.as_dict() for rule in trajectory.evaluate(records)]
return {
"generated": date.today().isoformat(),
"session": session,
"trace": trace_summary(records),
"structure": structural_score(kb_dir, report),
"trajectory": rules,
"violations": [r for r in rules if not r["passed"] and r["severity"] == "error"],
}
def failed(scorecard: dict) -> bool:
"""What makes a run a failure: a broken tree, or a violated invariant.
Advisories never fail a run. A trace that recorded nothing does not fail one
either - a session that used no tools is not a session that misbehaved.
"""
return bool(scorecard["structure"]["hard_errors"] or scorecard["violations"])
def render_markdown(scorecard: dict) -> str:
trace = scorecard["trace"]
lines = [
f"# Eval Score - {scorecard['session']} ({scorecard['generated']})",
"",
f"**{'FAILED' if failed(scorecard) else 'passed'}** - "
f"{trace['events']} event(s) from {', '.join(trace['sources']) or 'no source'}.",
"",
"## Trajectory (L2)",
"",
]
for rule in scorecard["trajectory"]:
if rule.get("skipped"):
mark = "skip"
elif rule["passed"]:
mark = "ok "
else:
mark = "FAIL" if rule["severity"] == "error" else "warn"
lines.append(f"- `{mark}` **{rule['id']}** - {rule['description']}")
if rule.get("skipped") and rule.get("skip_reason"):
lines.append(f" - {rule['skip_reason']}")
for finding in rule["findings"]:
detail = ", ".join(f"{k}={v}" for k, v in finding.items() if k != "ts")
lines.append(f" - {finding.get('ts', '')} {detail}")
lines += ["", "## Structure (L1)", ""]
structure = scorecard["structure"]
lines.append(f"{structure['page_count']} page(s); "
f"hard errors: {'yes' if structure['hard_errors'] else 'no'}.")
lines.append("")
for label, group in (("Errors", "errors"), ("Advisories", "advisories")):
found = {k: v for k, v in structure[group].items() if v}
lines.append(f"**{label}:** " + (", ".join(f"{k}={v}" for k, v in found.items())
if found else "none"))
lines += ["", "## Trace", ""]
for event, count in trace["by_event"].items():
lines.append(f"- `{event}`: {count}")
if trace["completeness"]:
lines += ["", "Reportable by this session's harness(es): "
+ ", ".join(f"`{c}`" for c in trace["completeness"])]
return "\n".join(lines)