Files changed: - .gitattributes - CHANGES.md - VERSION - raw/CONTRACT.md - tools/README.md - tools/chemenu/commands/_util.py - tools/chemenu/commands/dist_cmd.py - tools/chemenu/commands/docs_verify.py - tools/chemenu/commands/doctor.py - tools/chemenu/commands/eval_cmd.py - tools/chemenu/commands/git_publish.py - tools/chemenu/commands/index_build.py - tools/chemenu/commands/lint.py - tools/chemenu/commands/log_append.py - tools/chemenu/commands/migrate_cmd.py - tools/chemenu/commands/provenance_cmd.py - tools/chemenu/commands/raw_cmd.py - tools/chemenu/commands/run_budget.py - tools/chemenu/commands/upstream_cmd.py - tools/chemenu/commands/version_cmd.py - tools/chemenu/commands/work_cmd.py - tools/chemenu/config.py - tools/chemenu/corpus_cache.py - tools/chemenu/filelock.py - tools/chemenu/frontmatter_io.py - tools/chemenu/kb_scan.py - tools/chemenu/kb_state.py - tools/chemenu/lint_core.py - tools/chemenu/prerequisites.py - tools/chemenu/provenance.py - tools/chemenu/search/base.py - tools/chemenu/search/ripgrep.py - tools/chemenu/telemetry/writer.py - tools/chemenu/tests/test_dist_cmd.py - tools/chemenu/tests/test_portability.py - tools/chemenu/tests/test_search.py - tools/chemenu/tests/test_trace_ingest.py - tools/chemenu/type_resolver.py - tools/chemenu/upload.py - tools/chemenu/version.py - tools/run_wikitool.py - tools/trace_ingest.py
173 lines
6.5 KiB
Python
173 lines
6.5 KiB
Python
"""`wikitool eval` - score what a session did against what it left behind.
|
|
|
|
Read-only over `kb/`: the command runs lint's checks in-process and reads a
|
|
trace. It writes only into `reports/evals/`, which is gitignored like the rest of
|
|
that stage.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from datetime import date
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
import typer
|
|
|
|
from chemenu import cli_contract, config
|
|
from chemenu.commands._util import fail, rel_path, success
|
|
from chemenu.evals import scorecard
|
|
from chemenu.session import session_id as current_session_id
|
|
from chemenu.telemetry import reader
|
|
from chemenu.telemetry.writer import trace_root
|
|
|
|
app = typer.Typer(help="Score a traced session (see EVALS.md).")
|
|
|
|
EVALS_DIR = config.REPORTS_DIR / "evals"
|
|
|
|
|
|
@app.command("sessions")
|
|
@cli_contract.record(cli_contract.CommandRecord(
|
|
path="eval sessions",
|
|
summary="List the sessions that have a trace under `reports/telemetry/`.",
|
|
synopsis=(cli_contract.Variant(usage="eval sessions [--json]"),),
|
|
properties=cli_contract.Properties(
|
|
effect=cli_contract.Effect.READ,
|
|
idempotent=cli_contract.Idempotent.YES,
|
|
atomic="Read-only",
|
|
budget=cli_contract.Budget.EXEMPT,
|
|
),
|
|
notes=(
|
|
"Lists the sessions that have a trace under `reports/telemetry/`, most recent first.",
|
|
"Never fails; an empty list is a valid answer.",
|
|
"Read-only and exempt from the Iteration Budget Gate.",
|
|
),
|
|
failures=(),
|
|
examples=(
|
|
"tools/wikitool eval sessions",
|
|
"tools/wikitool eval sessions --json",
|
|
),
|
|
see_also=(
|
|
"`wikitool eval score` - scores one of them",
|
|
"`EVALS.md` - how telemetry and evaluation work",
|
|
),
|
|
))
|
|
def sessions_command(
|
|
json_out: bool = typer.Option(False, "--json", help="Print the list as JSON"),
|
|
):
|
|
"""List sessions that have a trace, most recent first."""
|
|
found = reader.sessions()
|
|
if json_out:
|
|
typer.echo(json.dumps(found, indent=2))
|
|
return
|
|
if not found:
|
|
success(f"No traces yet under {rel_path(trace_root())}.")
|
|
return
|
|
for name in found:
|
|
records = reader.read_trace(name)
|
|
first = records[0]["ts"][:19] if records else "-"
|
|
typer.echo(f"{name:40} {len(records):5d} event(s) since {first}")
|
|
|
|
|
|
@app.command("score")
|
|
@cli_contract.record(cli_contract.CommandRecord(
|
|
path="eval score",
|
|
summary="Score one traced session.",
|
|
synopsis=(cli_contract.Variant(
|
|
usage="eval score [--session <id>] [--json] [--markdown out.md] [--save] "
|
|
"[--fail-on-error]",
|
|
),),
|
|
properties=cli_contract.Properties(
|
|
effect=cli_contract.Effect.READ,
|
|
idempotent=cli_contract.Idempotent.YES,
|
|
atomic="Read-only, apart from the files `--save`/`--markdown` write",
|
|
budget=cli_contract.Budget.EXEMPT,
|
|
),
|
|
notes=(
|
|
"Scores one traced session: structural state from `lint`'s own checks (L1) plus "
|
|
"trajectory rules over the trace (L2) - was a refused call repeated unchanged, was a "
|
|
"gate flag passed without that gate having refused anything, did a publish of `kb/` "
|
|
"pages go unlogged.",
|
|
"Defaults to the current session; `--session <id>` picks another.",
|
|
"`--save` writes `reports/evals/<date>/<session>.{json,md}`; `--markdown` writes the "
|
|
"report to the named file.",
|
|
"A session records nothing when telemetry is off - `WIKI_TRACE=0`, or a distributed "
|
|
"instance with no opt-in (`wikitool doctor` says which) - so an absent trace is not "
|
|
"necessarily a fault.",
|
|
"Read-only over `kb/`, safe to retry, and exempt from the Iteration Budget Gate.",
|
|
),
|
|
failures=(
|
|
cli_contract.Failure(
|
|
cause="No trace exists for the named session",
|
|
reaction="Run `eval sessions` to see which ids exist; check with `doctor` whether "
|
|
"telemetry is on",
|
|
),
|
|
cli_contract.Failure(
|
|
cause="Only with `--fail-on-error`: the tree has hard errors or an invariant was "
|
|
"violated",
|
|
reaction="Act on the scorecard; re-run only to re-measure",
|
|
),
|
|
),
|
|
examples=(
|
|
"tools/wikitool eval score",
|
|
"tools/wikitool eval score --session wiki-1727330000 --save",
|
|
),
|
|
see_also=(
|
|
"`wikitool eval sessions` - which session ids exist",
|
|
"`EVALS.md` - the scoring levels",
|
|
),
|
|
))
|
|
def score_command(
|
|
session: Optional[str] = typer.Option(
|
|
None, "--session",
|
|
help="Session to score. Defaults to this shell's session, the same id the "
|
|
"budget gate uses.",
|
|
),
|
|
json_out: bool = typer.Option(False, "--json", help="Print the scorecard as JSON"),
|
|
markdown_out: Optional[Path] = typer.Option(
|
|
None, "--markdown",
|
|
help="Write a markdown scorecard here, conventionally under reports/evals/.",
|
|
),
|
|
save: bool = typer.Option(
|
|
False, "--save",
|
|
help="Write the scorecard to reports/evals/<date>/<session>.{json,md}.",
|
|
),
|
|
fail_on_error: bool = typer.Option(
|
|
False, "--fail-on-error",
|
|
help="Exit non-zero when the tree has hard errors or an invariant was violated.",
|
|
),
|
|
):
|
|
"""Score one session: structural state (L1) plus trajectory rules (L2)."""
|
|
target = session or current_session_id()
|
|
records = reader.read_trace(target)
|
|
if not records:
|
|
fail(
|
|
f"No trace for session '{target}'. `wikitool eval sessions` lists the ones "
|
|
"that exist; a session records nothing when WIKI_TRACE=0."
|
|
)
|
|
|
|
card = scorecard.score(target, records)
|
|
|
|
if save:
|
|
directory = EVALS_DIR / date.today().isoformat()
|
|
directory.mkdir(parents=True, exist_ok=True)
|
|
stem = target.replace("/", "__")
|
|
(directory / f"{stem}.json").write_text(
|
|
json.dumps(card, indent=2), encoding="utf-8", newline="\n"
|
|
)
|
|
(directory / f"{stem}.md").write_text(
|
|
scorecard.render_markdown(card) + "\n", encoding="utf-8", newline="\n"
|
|
)
|
|
success(f"Wrote {rel_path(directory / stem)}.json/.md")
|
|
if markdown_out:
|
|
markdown_out.write_text(
|
|
scorecard.render_markdown(card) + "\n", encoding="utf-8", newline="\n"
|
|
)
|
|
success(f"Wrote {rel_path(markdown_out)}")
|
|
if json_out:
|
|
typer.echo(json.dumps(card, indent=2))
|
|
if not json_out and not markdown_out and not save:
|
|
typer.echo(scorecard.render_markdown(card))
|
|
|
|
if fail_on_error and scorecard.failed(card):
|
|
raise typer.Exit(code=1)
|