Files
chemenu/tools/chemenu/evals/scorecard.py
T
torben 3916cb9541
CI / verify (push) Successful in 1m2s
Release / release (push) Successful in 37s
Link-Katalog: authored/alternative-to/addresses, entity→entity-Lineage, Lint-Befund redundant_see_also (4.7.0, #43 #49)
Files changed:
- CHANGES.md
- VERSION
- instructions/link-taxonomy.md
- kb/concepts/COLLECTION.md
- kb/entities/COLLECTION.md
- kb/entities/people/Andrej Karpathy.md
- kb/entities/people/Vannevar Bush.md
- tools/chemenu/evals/scorecard.py
- tools/chemenu/links.py
- tools/chemenu/lint_core.py
- tools/chemenu/tests/test_lint.py
2026-09-04 18:44:22 +02:00

129 lines
5.0 KiB
Python

"""Scoring a session: the structural state it left, and the path it took there.
Two levels, deliberately both hard-oracle:
- **L1, structure.** Counters from `lint`, which answers what the tree looks
like now. It is the same report `lint --fail-on-error` reads, so a score and a
lint run can never disagree.
- **L2, trajectory.** Rules over the trace, which answer how the tree got that
way. This is the half unit tests cannot reach.
There is no judge here and no rubric. Soft-oracle scoring waits until a failure
taxonomy exists - see EVALS.md on the phase gate.
"""
from __future__ import annotations
from datetime import date
from pathlib import Path
from chemenu import config
from chemenu.commands.lint import HARD_ERROR_KEYS, has_hard_errors, run_lint
from chemenu.evals import trajectory
from chemenu.telemetry import reader
# Advisory structural findings: real, but not a broken tree. Kept separate so a
# score reports them without failing on them.
ADVISORY_KEYS = (
"orphan_pages",
"quote_limit_violations",
"uncovered_raw_files",
"unmarked_provenance",
"missing_from_index",
"title_mismatches",
"redundant_see_also",
)
def structural_score(kb_dir: Path | None = None, report: dict | None = None) -> dict:
report = report if report is not None else run_lint(kb_dir or config.KB_DIR)
return {
"page_count": report["page_count"],
"hard_errors": has_hard_errors(report),
"errors": {key: len(report.get(key) or []) for key in HARD_ERROR_KEYS},
"advisories": {key: len(report.get(key) or []) for key in ADVISORY_KEYS},
}
def trace_summary(records: list[dict]) -> dict:
counts: dict[str, int] = {}
sources: list[str] = []
for record in records:
event = record.get("event", "?")
counts[event] = counts.get(event, 0) + 1
source = record.get("source")
if source and source not in sources:
sources.append(source)
return {
"events": len(records),
"sources": sources,
"by_event": dict(sorted(counts.items())),
"completeness": reader.completeness(records),
"first": records[0]["ts"] if records else None,
"last": records[-1]["ts"] if records else None,
}
def score(session: str, records: list[dict] | None = None,
kb_dir: Path | None = None, report: dict | None = None) -> dict:
records = reader.read_trace(session) if records is None else records
rules = [rule.as_dict() for rule in trajectory.evaluate(records)]
return {
"generated": date.today().isoformat(),
"session": session,
"trace": trace_summary(records),
"structure": structural_score(kb_dir, report),
"trajectory": rules,
"violations": [r for r in rules if not r["passed"] and r["severity"] == "error"],
}
def failed(scorecard: dict) -> bool:
"""What makes a run a failure: a broken tree, or a violated invariant.
Advisories never fail a run. A trace that recorded nothing does not fail one
either - a session that used no tools is not a session that misbehaved.
"""
return bool(scorecard["structure"]["hard_errors"] or scorecard["violations"])
def render_markdown(scorecard: dict) -> str:
trace = scorecard["trace"]
lines = [
f"# Eval Score - {scorecard['session']} ({scorecard['generated']})",
"",
f"**{'FAILED' if failed(scorecard) else 'passed'}** - "
f"{trace['events']} event(s) from {', '.join(trace['sources']) or 'no source'}.",
"",
"## Trajectory (L2)",
"",
]
for rule in scorecard["trajectory"]:
if rule.get("skipped"):
mark = "skip"
elif rule["passed"]:
mark = "ok "
else:
mark = "FAIL" if rule["severity"] == "error" else "warn"
lines.append(f"- `{mark}` **{rule['id']}** - {rule['description']}")
if rule.get("skipped") and rule.get("skip_reason"):
lines.append(f" - {rule['skip_reason']}")
for finding in rule["findings"]:
detail = ", ".join(f"{k}={v}" for k, v in finding.items() if k != "ts")
lines.append(f" - {finding.get('ts', '')} {detail}")
lines += ["", "## Structure (L1)", ""]
structure = scorecard["structure"]
lines.append(f"{structure['page_count']} page(s); "
f"hard errors: {'yes' if structure['hard_errors'] else 'no'}.")
lines.append("")
for label, group in (("Errors", "errors"), ("Advisories", "advisories")):
found = {k: v for k, v in structure[group].items() if v}
lines.append(f"**{label}:** " + (", ".join(f"{k}={v}" for k, v in found.items())
if found else "none"))
lines += ["", "## Trace", ""]
for event, count in trace["by_event"].items():
lines.append(f"- `{event}`: {count}")
if trace["completeness"]:
lines += ["", "Reportable by this session's harness(es): "
+ ", ".join(f"`{c}`" for c in trace["completeness"])]
return "\n".join(lines)