Files changed: - CHANGES.md - VERSION - instructions/link-taxonomy.md - kb/concepts/COLLECTION.md - kb/entities/COLLECTION.md - kb/entities/people/Andrej Karpathy.md - kb/entities/people/Vannevar Bush.md - tools/chemenu/evals/scorecard.py - tools/chemenu/links.py - tools/chemenu/lint_core.py - tools/chemenu/tests/test_lint.py
129 lines
5.0 KiB
Python
129 lines
5.0 KiB
Python
"""Scoring a session: the structural state it left, and the path it took there.
|
|
|
|
Two levels, deliberately both hard-oracle:
|
|
|
|
- **L1, structure.** Counters from `lint`, which answers what the tree looks
|
|
like now. It is the same report `lint --fail-on-error` reads, so a score and a
|
|
lint run can never disagree.
|
|
- **L2, trajectory.** Rules over the trace, which answer how the tree got that
|
|
way. This is the half unit tests cannot reach.
|
|
|
|
There is no judge here and no rubric. Soft-oracle scoring waits until a failure
|
|
taxonomy exists - see EVALS.md on the phase gate.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from datetime import date
|
|
from pathlib import Path
|
|
|
|
from chemenu import config
|
|
from chemenu.commands.lint import HARD_ERROR_KEYS, has_hard_errors, run_lint
|
|
from chemenu.evals import trajectory
|
|
from chemenu.telemetry import reader
|
|
|
|
# Advisory structural findings: real, but not a broken tree. Kept separate so a
|
|
# score reports them without failing on them.
|
|
ADVISORY_KEYS = (
|
|
"orphan_pages",
|
|
"quote_limit_violations",
|
|
"uncovered_raw_files",
|
|
"unmarked_provenance",
|
|
"missing_from_index",
|
|
"title_mismatches",
|
|
"redundant_see_also",
|
|
)
|
|
|
|
|
|
def structural_score(kb_dir: Path | None = None, report: dict | None = None) -> dict:
|
|
report = report if report is not None else run_lint(kb_dir or config.KB_DIR)
|
|
return {
|
|
"page_count": report["page_count"],
|
|
"hard_errors": has_hard_errors(report),
|
|
"errors": {key: len(report.get(key) or []) for key in HARD_ERROR_KEYS},
|
|
"advisories": {key: len(report.get(key) or []) for key in ADVISORY_KEYS},
|
|
}
|
|
|
|
|
|
def trace_summary(records: list[dict]) -> dict:
|
|
counts: dict[str, int] = {}
|
|
sources: list[str] = []
|
|
for record in records:
|
|
event = record.get("event", "?")
|
|
counts[event] = counts.get(event, 0) + 1
|
|
source = record.get("source")
|
|
if source and source not in sources:
|
|
sources.append(source)
|
|
return {
|
|
"events": len(records),
|
|
"sources": sources,
|
|
"by_event": dict(sorted(counts.items())),
|
|
"completeness": reader.completeness(records),
|
|
"first": records[0]["ts"] if records else None,
|
|
"last": records[-1]["ts"] if records else None,
|
|
}
|
|
|
|
|
|
def score(session: str, records: list[dict] | None = None,
|
|
kb_dir: Path | None = None, report: dict | None = None) -> dict:
|
|
records = reader.read_trace(session) if records is None else records
|
|
rules = [rule.as_dict() for rule in trajectory.evaluate(records)]
|
|
return {
|
|
"generated": date.today().isoformat(),
|
|
"session": session,
|
|
"trace": trace_summary(records),
|
|
"structure": structural_score(kb_dir, report),
|
|
"trajectory": rules,
|
|
"violations": [r for r in rules if not r["passed"] and r["severity"] == "error"],
|
|
}
|
|
|
|
|
|
def failed(scorecard: dict) -> bool:
|
|
"""What makes a run a failure: a broken tree, or a violated invariant.
|
|
|
|
Advisories never fail a run. A trace that recorded nothing does not fail one
|
|
either - a session that used no tools is not a session that misbehaved.
|
|
"""
|
|
return bool(scorecard["structure"]["hard_errors"] or scorecard["violations"])
|
|
|
|
|
|
def render_markdown(scorecard: dict) -> str:
|
|
trace = scorecard["trace"]
|
|
lines = [
|
|
f"# Eval Score - {scorecard['session']} ({scorecard['generated']})",
|
|
"",
|
|
f"**{'FAILED' if failed(scorecard) else 'passed'}** - "
|
|
f"{trace['events']} event(s) from {', '.join(trace['sources']) or 'no source'}.",
|
|
"",
|
|
"## Trajectory (L2)",
|
|
"",
|
|
]
|
|
for rule in scorecard["trajectory"]:
|
|
if rule.get("skipped"):
|
|
mark = "skip"
|
|
elif rule["passed"]:
|
|
mark = "ok "
|
|
else:
|
|
mark = "FAIL" if rule["severity"] == "error" else "warn"
|
|
lines.append(f"- `{mark}` **{rule['id']}** - {rule['description']}")
|
|
if rule.get("skipped") and rule.get("skip_reason"):
|
|
lines.append(f" - {rule['skip_reason']}")
|
|
for finding in rule["findings"]:
|
|
detail = ", ".join(f"{k}={v}" for k, v in finding.items() if k != "ts")
|
|
lines.append(f" - {finding.get('ts', '')} {detail}")
|
|
lines += ["", "## Structure (L1)", ""]
|
|
structure = scorecard["structure"]
|
|
lines.append(f"{structure['page_count']} page(s); "
|
|
f"hard errors: {'yes' if structure['hard_errors'] else 'no'}.")
|
|
lines.append("")
|
|
for label, group in (("Errors", "errors"), ("Advisories", "advisories")):
|
|
found = {k: v for k, v in structure[group].items() if v}
|
|
lines.append(f"**{label}:** " + (", ".join(f"{k}={v}" for k, v in found.items())
|
|
if found else "none"))
|
|
lines += ["", "## Trace", ""]
|
|
for event, count in trace["by_event"].items():
|
|
lines.append(f"- `{event}`: {count}")
|
|
if trace["completeness"]:
|
|
lines += ["", "Reportable by this session's harness(es): "
|
|
+ ", ".join(f"`{c}`" for c in trace["completeness"])]
|
|
return "\n".join(lines)
|