Files
chemenu/tools/chemenu/commands/_util.py
T
torben 04aebdeccf
CI / verify (push) Successful in 2m9s
Release / release (push) Successful in 38s
feat: path budget - a file's path stays at 160 characters or fewer; new, rename, move and raw accept refuse more, lint reports Long Paths (#163)
Files changed:
- CHANGES.md
- README.md
- VERSION
- instructions/page-lifecycle.md
- kb/CONTRACT.md
- tools/CONTRACT.md
- tools/chemenu/commands/_util.py
- tools/chemenu/commands/lint.py
- tools/chemenu/commands/new_page.py
- tools/chemenu/commands/page_ops.py
- tools/chemenu/commands/raw_cmd.py
- tools/chemenu/lint_core.py
- tools/chemenu/tests/test_lint.py
- tools/chemenu/tests/test_new_page.py
- tools/chemenu/tests/test_page_ops.py
- tools/chemenu/tests/test_raw_cmd.py
- tools/chemenu/tests/test_titles.py
- tools/chemenu/titles.py
2026-09-30 23:08:36 +02:00

311 lines
12 KiB
Python

"""Shared helpers for wikitool subcommands."""
from __future__ import annotations
import os
import re
import sys
from datetime import date
from pathlib import Path
from typing import Any, Dict, Optional
import typer
from rich.console import Console
from rich.markup import escape
console = Console()
# A third outcome alongside success (0) and validation error (1): the command
# is refusing until a *human* has seen its output and cleared it. It exists as
# its own code so the caller - an agent, a harness hook, a CI job, a trajectory
# scorer - can tell "stop and ask the user" apart from "your input was wrong,
# fix it and retry". Nothing about *why* clearance is needed lives in the agent
# instructions: the command's own output carries the reason, the evidence, and
# the exact re-run line.
EXIT_NEEDS_CLEARANCE = 42
def success(msg: str) -> None:
console.print(f"[green]OK[/green] {msg}")
# Set by `fail()`, read once per process by the CLI entry point. Exit 1 raised
# through `fail()` means the command declined and did the thing it was asked
# for: the argument was rejected, or a read-only check reported findings.
# Neither is an iteration step on the wiki, so the Iteration Budget Gate gives
# the slot back (see run_budget.refund). A command that has already done its
# work and then reports a non-zero result - `lint --fail-on-error` writes its
# report first - raises `typer.Exit(1)` directly and stays counted.
_declined = False
def declined() -> bool:
"""Whether this process left through `fail()`."""
return _declined
def fail(msg: str) -> None:
"""Print `ERROR <msg>` and leave through `typer.Exit(1)`.
Followed by the command's ON FAILURE hint on stderr - see
`_print_failure_hint`."""
global _declined
_declined = True
console.print(f"[bold red]ERROR[/bold red] {msg}")
_print_failure_hint()
raise typer.Exit(code=1)
def _print_failure_hint() -> None:
"""Print the running command's ON FAILURE hint to stderr, right after
the `ERROR` line `fail()` just wrote to stdout (Gitea #143): an agent
sees the reaction without a second `wikitool <path> -h` call. Plain text,
not through `console` - a reaction can contain a literal `[--flag]`,
which Rich would otherwise try to read as markup.
Reaches into `typer._click`, the same private, unpinned module `cli.py`
patches for its help rendering (see its own comment on why this is safe
to do). A failure here - no Click context yet (`fail()` called from
outside a command, see Gitea #147), a future typer that restructures the
module, a record this path cannot resolve - must not turn a validation
error into a crash: it is swallowed, and the call prints only the
`ERROR` line, exactly as it did before this hint existed.
"""
console.file.flush()
try:
from typer._click.globals import get_current_context
from chemenu import cli_contract
ctx = get_current_context(silent=True)
if ctx is None:
return
record = cli_contract.get(cli_contract.path_of(ctx))
if record is None:
return
sys.stderr.write(cli_contract.render_failure_hint(record) + "\n")
sys.stderr.flush()
except Exception:
pass
def needs_clearance(msg: str) -> None:
"""Refuse with EXIT_NEEDS_CLEARANCE. The message is written to be shown to
a human verbatim - it is the whole user-facing artifact of this gate."""
console.print(f"[bold yellow]NEEDS USER CLEARANCE[/bold yellow] {msg}")
raise typer.Exit(code=EXIT_NEEDS_CLEARANCE)
# A comma preceded by a backslash is a literal comma, not a separator.
_UNESCAPED_COMMA = re.compile(r"(?<!\\),")
def parse_list(value: str | None) -> list[str]:
"""Split a comma-separated CLI value into list elements.
`\\,` is an escaped literal comma: it survives the split and lands inside
the element. Without it a list format simply cannot express an element
that contains a comma - and shell quoting is no help, because the quotes
are gone long before this sees the string. Paths and page titles carry
commas often enough for that to matter: it once cost a `raw/` file its
original name, which `raw/CONTRACT.md` forbids.
"""
if not value:
return []
parts = (part.replace("\\,", ",").strip() for part in _UNESCAPED_COMMA.split(value))
return [part for part in parts if part]
def coerce_set_value(raw_value: str, field_schema: Optional[Dict[str, Any]]) -> Any:
"""Coerce a `--set field=value` string to the type its schema declares.
Arrays are comma-split (see `parse_list` for the escape), numbers are
parsed as float/int, booleans as true/false; everything else stays a
string. Unknown fields (no schema entry) pass through as strings and are
then caught by schema validation's `additionalProperties: false`.
"""
declared = (field_schema or {}).get("type")
if declared == "array":
return parse_list(raw_value)
if declared == "number":
try:
return float(raw_value)
except ValueError:
return raw_value
if declared == "integer":
try:
return int(raw_value)
except ValueError:
return raw_value
if declared == "boolean":
if raw_value.lower() in ("true", "false"):
return raw_value.lower() == "true"
return raw_value
def parse_set_fields(
set_fields: Optional[list[str]], schema: Optional[Dict[str, Any]], flag: str = "--set"
) -> Dict[str, Any]:
"""Parse repeated `<flag> field=value` pairs into a frontmatter dict,
coercing each value by the field's declared schema type.
Repeating the flag for an *array* field appends rather than replaces, so
`--set raw_files=a --set raw_files=b` yields both. That is the form that
needs no separator at all, and therefore the one to reach for when an
element contains a comma; `\\,` inside a single value does the same job
for a one-liner. Repeating a scalar field still means "last one wins" -
there is nothing to append to.
Note that this is per *invocation*. What a parsed value then means for a
page already on disk is the caller's decision: `new` writes it as the
page's initial value, while `touch` replaces, extends or subtracts
depending on which flag it came from.
"""
explicit: Dict[str, Any] = {}
properties = (schema or {}).get("properties", {})
for pair in set_fields or []:
if "=" not in pair:
fail(f"{flag} expects field=value, got: {pair}")
field_name, raw_value = pair.split("=", 1)
field_name = field_name.strip()
if not field_name:
fail(f"{flag} expects field=value, got: {pair}")
value = coerce_set_value(raw_value, properties.get(field_name))
previous = explicit.get(field_name)
if isinstance(value, list) and isinstance(previous, list):
previous.extend(value)
else:
explicit[field_name] = value
return explicit
def check_raw_files_exist(raw_files: Any) -> None:
"""Verify every `raw_files:` entry is an existing file.
This is the one validation that genuinely cannot live in the schema:
it is filesystem I/O, not a data-shape constraint. Cardinality
(`minItems: 1`) is already enforced by the schema itself, so only
existence and file-vs-directory are checked here.
Shared by `new` and `touch` - both write the field, and a page pointing at
a raw file that is not there is the same defect whichever wrote it.
"""
from chemenu import config
for raw_path in raw_files or []:
full_path = config.ROOT / raw_path
if not full_path.exists():
fail(
f"raw_files path does not exist: {raw_path}\n"
" This is one element after splitting the value on commas. If the real "
"filename contains a comma, escape it as `\\,` or pass one `--set "
"raw_files=<path>` per file - never rename the raw file to fit the flag."
)
if full_path.is_dir():
fail(f"raw_files must be a file, not a directory: {raw_path}")
def today_iso() -> str:
return date.today().isoformat()
def rel_path(path: Path) -> str:
"""Format a path relative to the repo root for display, falling back to
the raw path if it lies outside the root (e.g. in tests)."""
from chemenu import config
try:
return str(Path(path).relative_to(config.ROOT))
except ValueError:
return str(path)
def check_title(name: str) -> None:
"""Fail if `name` cannot be a page title (see `chemenu.titles`).
A title is a file name, so this runs on every platform and for every type,
whether or not the page lands under `kb/`.
"""
from chemenu.titles import title_problems
problems = title_problems(name)
if problems:
fail(escape(f"'{name}' cannot be a page title: " + "; ".join(problems) + "."))
def path_budget_problem_for(path: Path) -> str | None:
"""`chemenu.titles.path_budget_problem` for a path on disk, measured from the
instance root."""
from chemenu.titles import path_budget_problem
return path_budget_problem(rel_path(path).replace(os.sep, "/"))
def check_path_budget(path: Path, remedy: str) -> None:
"""Fail if `path` is over the path budget (see `chemenu.titles.PATH_BUDGET`).
Runs before any write, for every root. `remedy` is the sentence that tells
the caller what to shorten, since the name comes from a title in one command
and from a file in `incoming/` in another.
"""
problem = path_budget_problem_for(path)
if problem:
fail(escape(f"Cannot write {problem}. {remedy}"))
def check_collision(name: str, *, ignore: Path | None = None) -> None:
"""Fail if a page under kb/ already has a title that collides with `name`.
The stem *is* the page title and wikilinks resolve by title alone, so two
files sharing a stem in different directories are indistinguishable to
every link in the wiki. Titles that differ only by case or Unicode
normalization collide as well, because NTFS and APFS fold them into one
file. `ignore` is the page being renamed, which may only change its case.
Shared by `new` and `rename`.
"""
from chemenu import config
from chemenu.kb_scan import iter_kb_pages
from chemenu.titles import collision_key
wanted = collision_key(name)
for path in iter_kb_pages(config.KB_DIR):
if path == ignore or collision_key(path.stem) != wanted:
continue
note = "" if path.stem == name else " (titles are compared without regard to case or Unicode normalization)"
fail(escape(
f"A page titled '{path.stem}' already exists at {rel_path(path)}, which collides "
f"with '{name}'{note}"
))
def target_conflict(path: Path, *, ignore: Path | None = None) -> Path | None:
"""The existing entry in `path`'s directory that `path` would clash with,
or None. Compared by `collision_key`, so it does not depend on the file
system the check happens to run on."""
from chemenu.titles import collision_key
if not path.parent.is_dir():
return None
wanted = collision_key(path.name)
for entry in sorted(path.parent.iterdir()):
if entry == ignore:
continue
if collision_key(entry.name) == wanted:
return entry
return None
def check_target_free(path: Path, *, ignore: Path | None = None) -> None:
"""Fail if writing `path` would overwrite, or land beside, an existing entry
that a case-insensitive file system would treat as the same file.
Holds for every root: `check_collision` only sees pages under kb/, so it
could not stop `new instruction --name gates` from overwriting
`instructions/gates.md`.
"""
clash = target_conflict(path, ignore=ignore)
if clash is not None:
fail(escape(
f"Cannot write {rel_path(path)}: {rel_path(clash)} already exists there "
"(names are compared without regard to case or Unicode normalization)."
))