Files changed: - .gitea/workflows/ci.yml - CHANGES.md - INSTALL.md - VERSION - instructions/bug-report.md - instructions/gates.md - instructions/setup-instance.md - instructions/upgrade-instance.md - reports/CONTRACT.md - tools/README.md - tools/bugreport.py - tools/chemenu/tests/test_bugreport.py
943 lines
35 KiB
Python
943 lines
35 KiB
Python
#!/usr/bin/env python3
|
|
"""Collect a bug-report bundle from this checkout.
|
|
|
|
Runs on the base Python with the standard library only, and imports nothing from
|
|
`chemenu`: the case it exists for is a checkout where `wikitool` does not start
|
|
- no venv, a broken package, a Python that is too old. Syntax stays at Python
|
|
3.8 so that even an old interpreter can still produce a report.
|
|
|
|
python tools/bugreport.py [--chronology FILE] [--transcript FILE]...
|
|
[--session ID | --no-trace] [--titles]
|
|
[--root DIR] [--out DIR]
|
|
|
|
The bundle is `<out>/bugreport-<UTC stamp>/` plus a zip beside it, in four
|
|
layers: the environment, the stack, what `wikitool` prints (if it starts), and
|
|
the agent's chronology. Secrets are always removed. Page titles are kept out of
|
|
everything this script generates unless `--titles` is given. Nothing is uploaded.
|
|
|
|
Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script
|
|
opens no gate. See instructions/bug-report.md.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import datetime
|
|
import json
|
|
import os
|
|
import platform
|
|
import re
|
|
import shutil
|
|
import subprocess
|
|
import sys
|
|
import unicodedata
|
|
import zipfile
|
|
from pathlib import Path
|
|
|
|
TIMEOUT_SECONDS = 300
|
|
PROBE_TIMEOUT_SECONDS = 20
|
|
REMOVED = "<removed>"
|
|
NOT_COLLECTED = "<not collected>"
|
|
SESSION_ENV = "WIKITOOL_SESSION_ID"
|
|
HARNESS_SESSION_ENV = "CLAUDE_CODE_SESSION_ID"
|
|
|
|
SECRET_NAME = re.compile(r"token|password|secret|key|auth", re.IGNORECASE)
|
|
|
|
HARNESS_MARKERS = (
|
|
(HARNESS_SESSION_ENV, "claude-code"),
|
|
)
|
|
|
|
ENV_ALLOW_EXACT = frozenset(
|
|
{
|
|
"PATH", "PATHEXT", "SHELL", "TERM", "TERM_PROGRAM", "TERM_PROGRAM_VERSION",
|
|
"COLORTERM", "LANG", "LANGUAGE", "VIRTUAL_ENV", "MSYSTEM", "COMSPEC",
|
|
"PSMODULEPATH",
|
|
}
|
|
| {name for name, _ in HARNESS_MARKERS}
|
|
)
|
|
ENV_ALLOW_PREFIX = ("LC_", "PYTHON", "WIKI_", "WIKITOOL_", "CHEMENU_")
|
|
|
|
# Names under kb/ that the stack generates or owns: not titles, and unreadable
|
|
# in a bundle if they are masked.
|
|
STACK_NAMES = frozenset(
|
|
{
|
|
"CONTRACT.md", "CONVENTIONS.md", "CONVENTIONS.md.template", "COLLECTION.md",
|
|
"COLLECTION.md.template", "INDEX.md", "index.md", "log.md", "provenance.md",
|
|
".gitkeep",
|
|
}
|
|
)
|
|
CONTENT_DIRS = ("kb", "raw")
|
|
CONTENT_COMMIT_DIRS = ("kb", "raw", "work")
|
|
|
|
MAY_CONTAIN_CONTENT = "may contain page content and titles"
|
|
|
|
PRIVACY_NOTICE = (
|
|
"This bundle is not pseudonymised. It contains private data: machine, user and path "
|
|
"names, PATH entries, git remotes and commit subjects and - if included - the session "
|
|
"trace, the chronology and transcripts, which may contain page content and titles. "
|
|
"Secrets are removed, but read the bundle before you share it. Choose the channel "
|
|
"yourself; neither this tool nor the agent uploads anything."
|
|
)
|
|
|
|
WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE)
|
|
WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]')
|
|
WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]")
|
|
URL_USERINFO = re.compile(r"(?P<scheme>[A-Za-z][A-Za-z0-9+.\-]*)://(?P<userinfo>[^/\s@\"'<>]+)@")
|
|
AUTH_HEADER = re.compile(r"(?im)(authorization\s*[:=]\s*)(?:(?:bearer|basic|token)\s+)?\S+")
|
|
BEARER = re.compile(r"(?i)\b(bearer)\s+[A-Za-z0-9._~+/=\-]{8,}")
|
|
|
|
|
|
# --------------------------------------------------------------------------- secrets
|
|
|
|
|
|
class Redactor:
|
|
"""Every secret value found while collecting, replaced again in a final pass."""
|
|
|
|
def __init__(self) -> None:
|
|
self.values = set()
|
|
|
|
def add(self, value) -> None:
|
|
if isinstance(value, str) and len(value) >= 4:
|
|
self.values.add(value)
|
|
|
|
def add_any(self, node) -> None:
|
|
if isinstance(node, dict):
|
|
for item in node.values():
|
|
self.add_any(item)
|
|
elif isinstance(node, (list, tuple)):
|
|
for item in node:
|
|
self.add_any(item)
|
|
else:
|
|
self.add(node)
|
|
|
|
def scrub_urls(self, text: str) -> str:
|
|
def repl(match):
|
|
scheme, userinfo = match.group("scheme"), match.group("userinfo")
|
|
if scheme.lower() in ("http", "https"):
|
|
self.add(userinfo)
|
|
for part in userinfo.split(":"):
|
|
self.add(part)
|
|
return scheme + "://"
|
|
if ":" in userinfo:
|
|
user, password = userinfo.split(":", 1)
|
|
self.add(password)
|
|
return scheme + "://" + user + "@"
|
|
return match.group(0)
|
|
|
|
return URL_USERINFO.sub(repl, text)
|
|
|
|
def scrub(self, text: str) -> str:
|
|
text = self.scrub_urls(text)
|
|
text = AUTH_HEADER.sub(lambda m: m.group(1) + REMOVED, text)
|
|
text = BEARER.sub(lambda m: m.group(1) + " " + REMOVED, text)
|
|
forms = set()
|
|
for value in self.values:
|
|
forms.add(value)
|
|
escaped = json.dumps(value, ensure_ascii=False)[1:-1]
|
|
if escaped != value:
|
|
forms.add(escaped)
|
|
for form in sorted(forms, key=len, reverse=True):
|
|
text = text.replace(form, REMOVED)
|
|
return text
|
|
|
|
def clean_json(self, node):
|
|
if isinstance(node, dict):
|
|
out = {}
|
|
for key, value in node.items():
|
|
if SECRET_NAME.search(str(key)):
|
|
self.add_any(value)
|
|
out[key] = REMOVED
|
|
else:
|
|
out[key] = self.clean_json(value)
|
|
return out
|
|
if isinstance(node, list):
|
|
return [self.clean_json(item) for item in node]
|
|
if isinstance(node, str):
|
|
return self.scrub_urls(node)
|
|
return node
|
|
|
|
|
|
# ------------------------------------------------------------------------ title shield
|
|
|
|
|
|
def _shape(rel: str) -> str:
|
|
parts = rel.split("/")
|
|
name = parts[-1]
|
|
stem, dot, suffix = name.rpartition(".")
|
|
flags = ["depth=%d" % len(parts), "len=%d" % len(rel)]
|
|
if " " in rel:
|
|
flags.append("space")
|
|
if any(ord(ch) > 127 for ch in rel):
|
|
flags.append("non-ascii")
|
|
return "%s/<%s>%s" % (parts[0], " ".join(flags), (dot + suffix) if dot and stem else "")
|
|
|
|
|
|
class TitleShield:
|
|
"""Replaces the relative path of a file under kb/ or raw/ with its shape.
|
|
|
|
A title that stands as bare prose is not found, on purpose: replacing titles
|
|
word by word would hit a title like "Git" everywhere in the bundle.
|
|
"""
|
|
|
|
def __init__(self, rel_paths) -> None:
|
|
self.forms = {}
|
|
for rel in rel_paths:
|
|
shape = _shape(rel)
|
|
for form in (rel, rel.replace("/", "\\"), rel.replace("/", "\\\\")):
|
|
self.forms[form] = shape
|
|
self.lengths = sorted({len(form) for form in self.forms}, reverse=True)
|
|
self.start = re.compile(r"(?:%s)(?:/|\\)" % "|".join(CONTENT_DIRS))
|
|
|
|
def apply(self, text: str) -> str:
|
|
if self.forms:
|
|
out, last = [], 0
|
|
for match in self.start.finditer(text):
|
|
i = match.start()
|
|
if i < last:
|
|
continue
|
|
for length in self.lengths:
|
|
shape = self.forms.get(text[i:i + length])
|
|
if shape:
|
|
out.append(text[last:i])
|
|
out.append(shape)
|
|
last = i + length
|
|
break
|
|
out.append(text[last:])
|
|
text = "".join(out)
|
|
return WIKILINK.sub(lambda m: "[[<title %s>]]" % _title_flags(m.group(1)), text)
|
|
|
|
|
|
def _title_flags(title: str) -> str:
|
|
flags = ["len=%d" % len(title)]
|
|
if " " in title:
|
|
flags.append("space")
|
|
if any(ord(ch) > 127 for ch in title):
|
|
flags.append("non-ascii")
|
|
return " ".join(flags)
|
|
|
|
|
|
# ---------------------------------------------------------------------------- helpers
|
|
|
|
|
|
class Result:
|
|
def __init__(self, returncode, stdout, stderr, note=None) -> None:
|
|
self.returncode = returncode
|
|
self.stdout = stdout
|
|
self.stderr = stderr
|
|
self.note = note
|
|
|
|
@property
|
|
def ok(self) -> bool:
|
|
return self.returncode == 0
|
|
|
|
def first_line(self) -> str:
|
|
for text in (self.stderr, self.stdout, self.note or ""):
|
|
for line in text.splitlines():
|
|
if line.strip():
|
|
return line.strip()
|
|
return ""
|
|
|
|
|
|
def _decode(data) -> str:
|
|
if data is None:
|
|
return ""
|
|
if isinstance(data, str):
|
|
return data
|
|
return data.decode("utf-8", errors="replace")
|
|
|
|
|
|
def run(cmd, cwd, env=None, timeout=PROBE_TIMEOUT_SECONDS) -> Result:
|
|
try:
|
|
proc = subprocess.run(
|
|
cmd, cwd=str(cwd), env=env, stdin=subprocess.DEVNULL,
|
|
stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout,
|
|
)
|
|
except subprocess.TimeoutExpired as exc:
|
|
return Result(None, _decode(exc.stdout), _decode(exc.stderr),
|
|
"timed out after %d s" % timeout)
|
|
except OSError as exc:
|
|
return Result(None, "", "", "could not start: %s" % exc)
|
|
return Result(proc.returncode, _decode(proc.stdout), _decode(proc.stderr))
|
|
|
|
|
|
def read_text(path: Path) -> str:
|
|
with open(str(path), "r", encoding="utf-8", errors="replace", newline="") as handle:
|
|
return handle.read()
|
|
|
|
|
|
def write_text(path: Path, text: str) -> None:
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
with open(str(path), "w", encoding="utf-8", newline="\n") as handle:
|
|
handle.write(text)
|
|
|
|
|
|
def dump_json(data) -> str:
|
|
return json.dumps(data, indent=2, ensure_ascii=False) + "\n"
|
|
|
|
|
|
def guarded(fn, *args):
|
|
"""A broken layer must not take the whole report down."""
|
|
try:
|
|
return fn(*args)
|
|
except Exception as exc: # noqa: BLE001 - the report is the point
|
|
return {"collector_error": "%s: %s" % (type(exc).__name__, exc)}
|
|
|
|
|
|
def session_slug(value: str) -> str:
|
|
return re.sub(r"[^A-Za-z0-9._-]+", "__", value).strip("_") or "unknown"
|
|
|
|
|
|
# ------------------------------------------------------------------- layer 1: environment
|
|
|
|
|
|
def _version_line(cmd, cwd) -> str:
|
|
result = run(cmd, cwd)
|
|
return result.first_line() if result.ok or result.first_line() else (result.note or "")
|
|
|
|
|
|
def collect_environment(root: Path, red: Redactor) -> dict:
|
|
data = {
|
|
"os": {
|
|
"platform": platform.platform(),
|
|
"system": platform.system(),
|
|
"release": platform.release(),
|
|
"version": platform.version(),
|
|
"machine": platform.machine(),
|
|
},
|
|
"python": _collect_python(root),
|
|
"shells": _collect_shells(root),
|
|
"harness": _collect_harness(),
|
|
"path_entries": [
|
|
{"entry": entry, "exists": os.path.isdir(entry)}
|
|
for entry in os.environ.get("PATH", "").split(os.pathsep) if entry
|
|
],
|
|
"env": _collect_env(red),
|
|
"tools_config": _collect_tools_config(root, red),
|
|
"git": _collect_git_config(root, red),
|
|
"venv": _collect_venv(root, red),
|
|
"line_endings": _collect_line_endings(root),
|
|
"gitattributes": _read_optional(root / ".gitattributes", red),
|
|
"windows": _collect_windows(root),
|
|
}
|
|
return data
|
|
|
|
|
|
def _collect_python(root: Path) -> dict:
|
|
running = {
|
|
"executable": sys.executable,
|
|
"version": sys.version.replace("\n", " "),
|
|
"version_info": list(sys.version_info[:3]),
|
|
}
|
|
candidates = []
|
|
for name, extra in (("python3", []), ("python", []), ("py", ["-3"])):
|
|
path = shutil.which(name)
|
|
entry = {"name": " ".join([name] + extra), "path": path}
|
|
if path:
|
|
entry["version"] = _version_line([path] + extra + ["--version"], root)
|
|
entry["windows_store_alias"] = "windowsapps" in path.lower()
|
|
candidates.append(entry)
|
|
return {"running": running, "candidates": candidates}
|
|
|
|
|
|
def _collect_shells(root: Path) -> dict:
|
|
shells = {"SHELL": os.environ.get("SHELL"), "COMSPEC": os.environ.get("COMSPEC")}
|
|
for name in ("bash", "zsh", "fish"):
|
|
path = shutil.which(name)
|
|
shells[name] = {"path": path,
|
|
"version": _version_line([path, "--version"], root) if path else None}
|
|
for name in ("pwsh", "powershell"):
|
|
path = shutil.which(name)
|
|
shells[name] = {
|
|
"path": path,
|
|
"version": _version_line(
|
|
[path, "-NoProfile", "-Command", "$PSVersionTable.PSVersion.ToString()"], root
|
|
) if path else None,
|
|
}
|
|
return shells
|
|
|
|
|
|
def _collect_harness() -> dict:
|
|
detected = [
|
|
{"marker": name, "harness": harness}
|
|
for name, harness in HARNESS_MARKERS if os.environ.get(name)
|
|
]
|
|
if os.environ.get("TERM_PROGRAM") == "vscode":
|
|
detected.append({"marker": "TERM_PROGRAM=vscode", "harness": "vscode terminal"})
|
|
return {"detected": detected or "unrecognised"}
|
|
|
|
|
|
def _env_allowed(name: str) -> bool:
|
|
upper = name.upper()
|
|
return upper in ENV_ALLOW_EXACT or upper.startswith(ENV_ALLOW_PREFIX)
|
|
|
|
|
|
def _collect_env(red: Redactor) -> dict:
|
|
env = {}
|
|
for name in sorted(os.environ):
|
|
value = os.environ[name]
|
|
if SECRET_NAME.search(name):
|
|
red.add(value)
|
|
env[name] = REMOVED if _env_allowed(name) else NOT_COLLECTED
|
|
elif _env_allowed(name):
|
|
env[name] = value
|
|
else:
|
|
env[name] = NOT_COLLECTED
|
|
return env
|
|
|
|
|
|
def _collect_tools_config(root: Path, red: Redactor) -> dict:
|
|
path = root / ".wikitool-tools.json"
|
|
if not path.is_file():
|
|
return {"status": "absent"}
|
|
try:
|
|
parsed = json.loads(read_text(path))
|
|
except ValueError as exc:
|
|
return {"status": "unparseable", "error": str(exc)}
|
|
checks = []
|
|
|
|
def walk(node, where):
|
|
if isinstance(node, dict):
|
|
for key, value in node.items():
|
|
walk(value, where + [str(key)])
|
|
elif isinstance(node, list):
|
|
for index, value in enumerate(node):
|
|
walk(value, where + [str(index)])
|
|
elif isinstance(node, str) and os.path.isabs(node):
|
|
checks.append({"at": ".".join(where), "path": node, "exists": os.path.exists(node)})
|
|
|
|
walk(parsed, [])
|
|
return {"status": "present", "content": red.clean_json(parsed), "path_checks": checks}
|
|
|
|
|
|
def _collect_git_config(root: Path, red: Redactor) -> dict:
|
|
version = run(["git", "--version"], root)
|
|
if not version.ok:
|
|
return {"available": False, "error": version.note or version.first_line()}
|
|
config = {}
|
|
for key in ("core.autocrlf", "core.longpaths", "core.filemode"):
|
|
result = run(["git", "config", "--get", key], root)
|
|
config[key] = result.stdout.strip() if result.ok else "(unset)"
|
|
name = run(["git", "config", "--get", "user.name"], root)
|
|
config["user.name is set"] = bool(name.ok and name.stdout.strip())
|
|
remotes = run(["git", "remote", "-v"], root)
|
|
return {
|
|
"available": True,
|
|
"version": version.stdout.strip(),
|
|
"config": config,
|
|
"remotes": red.scrub_urls(remotes.stdout).splitlines() if remotes.ok
|
|
else ["(not a repository, or git failed: %s)" % remotes.first_line()],
|
|
}
|
|
|
|
|
|
def _collect_venv(root: Path, red: Redactor) -> dict:
|
|
venv = root / "tools" / ".venv"
|
|
if not venv.is_dir():
|
|
return {"present": False}
|
|
layout = ("bin" if (venv / "bin").is_dir() else None) or (
|
|
"Scripts" if (venv / "Scripts").is_dir() else None)
|
|
cfg = venv / "pyvenv.cfg"
|
|
return {
|
|
"present": True,
|
|
"layout": layout or "unknown",
|
|
"python_found": any(
|
|
(venv / sub / exe).exists()
|
|
for sub, exe in (("bin", "python"), ("Scripts", "python.exe"))
|
|
),
|
|
"pyvenv.cfg": read_text(cfg) if cfg.is_file() else None,
|
|
}
|
|
|
|
|
|
def _collect_line_endings(root: Path) -> dict:
|
|
tools = root / "tools"
|
|
files = [tools / "wikitool"] + (sorted(tools.glob("*.ps1")) if tools.is_dir() else [])
|
|
result = {}
|
|
for path in files:
|
|
if not path.is_file():
|
|
continue
|
|
raw = path.read_bytes()
|
|
crlf = raw.count(b"\r\n")
|
|
result[path.name] = {"crlf": crlf, "lf_only": raw.count(b"\n") - crlf}
|
|
return result
|
|
|
|
|
|
def _read_optional(path: Path, red: Redactor):
|
|
return red.scrub_urls(read_text(path)) if path.is_file() else None
|
|
|
|
|
|
def _collect_windows(root: Path) -> dict:
|
|
if sys.platform != "win32":
|
|
return {"applicable": False}
|
|
data = {"applicable": True}
|
|
try:
|
|
import winreg # noqa: PLC0415 - only exists on Windows
|
|
|
|
with winreg.OpenKey(
|
|
winreg.HKEY_LOCAL_MACHINE, r"SYSTEM\CurrentControlSet\Control\FileSystem"
|
|
) as key:
|
|
data["LongPathsEnabled"] = winreg.QueryValueEx(key, "LongPathsEnabled")[0]
|
|
except Exception as exc: # noqa: BLE001
|
|
data["LongPathsEnabled"] = "unreadable: %s" % exc
|
|
pwsh = shutil.which("pwsh")
|
|
if pwsh:
|
|
policy = run([pwsh, "-NoProfile", "-Command", "Get-ExecutionPolicy -List | Out-String"], root)
|
|
data["execution_policy"] = policy.stdout.strip() or policy.first_line()
|
|
else:
|
|
data["execution_policy"] = "pwsh not found"
|
|
data["powershell_5_present"] = bool(shutil.which("powershell"))
|
|
marked = []
|
|
tools = root / "tools"
|
|
if tools.is_dir():
|
|
for path in sorted(p for p in tools.iterdir() if p.is_file()):
|
|
try:
|
|
with open(str(path) + ":Zone.Identifier", "r", encoding="utf-8",
|
|
errors="replace") as stream:
|
|
marked.append({"file": path.name, "zone": stream.read().strip()})
|
|
except OSError:
|
|
pass
|
|
data["mark_of_the_web"] = marked
|
|
return data
|
|
|
|
|
|
# ------------------------------------------------------------------------ layer 2: stack
|
|
|
|
|
|
def scan_tree(root: Path, name: str):
|
|
"""Structure of kb/ or raw/, and the relative paths it holds. No content."""
|
|
top = root / name
|
|
if not top.is_dir():
|
|
return {"present": False}, []
|
|
files, dirs = [], 0
|
|
depth_histogram, longest_rel, longest_abs = {}, 0, 0
|
|
counts = {
|
|
"abs_path_over_240": 0, "abs_path_over_260": 0, "names_with_space": 0,
|
|
"names_non_ascii": 0, "names_windows_forbidden_chars": 0,
|
|
"names_trailing_dot_or_space": 0, "names_reserved_device": 0,
|
|
"names_not_nfc": 0, "case_collisions": 0,
|
|
}
|
|
|
|
def judge(entry: str) -> None:
|
|
if " " in entry:
|
|
counts["names_with_space"] += 1
|
|
if any(ord(ch) > 127 for ch in entry):
|
|
counts["names_non_ascii"] += 1
|
|
if WINDOWS_FORBIDDEN.search(entry):
|
|
counts["names_windows_forbidden_chars"] += 1
|
|
if entry.endswith((" ", ".")):
|
|
counts["names_trailing_dot_or_space"] += 1
|
|
if WINDOWS_RESERVED.match(entry):
|
|
counts["names_reserved_device"] += 1
|
|
if unicodedata.normalize("NFC", entry) != entry:
|
|
counts["names_not_nfc"] += 1
|
|
|
|
for current, subdirs, names in os.walk(str(top)):
|
|
entries = sorted(subdirs) + sorted(names)
|
|
seen = {}
|
|
for entry in entries:
|
|
seen.setdefault(entry.casefold(), []).append(entry)
|
|
judge(entry)
|
|
counts["case_collisions"] += sum(1 for group in seen.values() if len(group) > 1)
|
|
dirs += len(subdirs)
|
|
for entry in names:
|
|
absolute = os.path.join(current, entry)
|
|
rel = os.path.relpath(absolute, str(root)).replace(os.sep, "/")
|
|
files.append(rel)
|
|
depth = rel.count("/") + 1
|
|
depth_histogram[str(depth)] = depth_histogram.get(str(depth), 0) + 1
|
|
longest_rel = max(longest_rel, len(rel))
|
|
longest_abs = max(longest_abs, len(os.path.abspath(absolute)))
|
|
if len(os.path.abspath(absolute)) > 240:
|
|
counts["abs_path_over_240"] += 1
|
|
if len(os.path.abspath(absolute)) > 260:
|
|
counts["abs_path_over_260"] += 1
|
|
summary = {
|
|
"present": True, "files": len(files), "directories": dirs,
|
|
"depth_histogram": depth_histogram, "longest_relative_path": longest_rel,
|
|
"longest_absolute_path": longest_abs,
|
|
}
|
|
summary.update(counts)
|
|
return summary, files
|
|
|
|
|
|
def collect_stack(root: Path, red: Redactor) -> dict:
|
|
version = root / "VERSION"
|
|
configs = {}
|
|
for path in sorted(root.glob(".wikitool-*.json")):
|
|
if path.name == ".wikitool-tools.json":
|
|
continue
|
|
try:
|
|
configs[path.name] = red.clean_json(json.loads(read_text(path)))
|
|
except ValueError:
|
|
configs[path.name] = {"unparseable": True, "bytes": path.stat().st_size}
|
|
return {
|
|
"VERSION": read_text(version).strip() if version.is_file() else "absent",
|
|
"config_files": configs,
|
|
"git": _collect_git_state(root),
|
|
}
|
|
|
|
|
|
def _collect_git_state(root: Path) -> dict:
|
|
quiet = ["git", "-c", "core.quotepath=off"]
|
|
inside = run(quiet + ["rev-parse", "--is-inside-work-tree"], root)
|
|
if not inside.ok:
|
|
return {"repository": False, "error": inside.first_line()}
|
|
branch = run(quiet + ["rev-parse", "--abbrev-ref", "HEAD"], root)
|
|
status = run(quiet + ["status", "--porcelain", "-z"], root)
|
|
log = run(quiet + ["log", "-n", "20", "--name-only", "--format=%x1e%h%x1f%s"], root)
|
|
return {
|
|
"repository": True,
|
|
"branch": branch.stdout.strip() if branch.ok else branch.first_line(),
|
|
"status": _parse_status(status.stdout) if status.ok else [status.first_line()],
|
|
"log": _parse_log(log.stdout) if log.ok else [log.first_line()],
|
|
}
|
|
|
|
|
|
def _parse_status(raw: str) -> list:
|
|
entries = raw.split("\0")
|
|
lines, i = [], 0
|
|
while i < len(entries):
|
|
entry = entries[i]
|
|
i += 1
|
|
if len(entry) < 4:
|
|
continue
|
|
code, path = entry[:2], entry[3:]
|
|
if code[0] in "RC" and i < len(entries):
|
|
lines.append("%s %s (from %s)" % (code, path, entries[i]))
|
|
i += 1
|
|
else:
|
|
lines.append("%s %s" % (code, path))
|
|
return lines
|
|
|
|
|
|
def _parse_log(raw: str) -> list:
|
|
lines = []
|
|
for block in raw.split("\x1e"):
|
|
if not block.strip():
|
|
continue
|
|
head, _, rest = block.partition("\n")
|
|
sha, _, subject = head.partition("\x1f")
|
|
files = [line for line in rest.splitlines() if line.strip()]
|
|
if files and all(f.startswith(tuple(d + "/" for d in CONTENT_COMMIT_DIRS)) for f in files):
|
|
parts = []
|
|
for directory in CONTENT_COMMIT_DIRS:
|
|
count = sum(1 for f in files if f.startswith(directory + "/"))
|
|
if count:
|
|
parts.append("%s %d" % (directory, count))
|
|
subject = "<content commit: %s files>" % ", ".join(parts)
|
|
lines.append("%s %s" % (sha.strip(), subject.strip()))
|
|
return lines
|
|
|
|
|
|
# ---------------------------------------------------------------------- layer 3: wikitool
|
|
|
|
|
|
def launcher_command(root: Path):
|
|
tools = root / "tools"
|
|
if sys.platform == "win32":
|
|
ps1, sh = tools / "wikitool.ps1", tools / "wikitool"
|
|
pwsh = shutil.which("pwsh")
|
|
if ps1.is_file() and pwsh:
|
|
return [pwsh, "-NoProfile", "-File", str(ps1)]
|
|
bash = shutil.which("bash")
|
|
if sh.is_file() and bash:
|
|
return [bash, str(sh)]
|
|
return None
|
|
launcher = tools / "wikitool"
|
|
return [str(launcher)] if launcher.is_file() else None
|
|
|
|
|
|
def resolve_trace_session(explicit):
|
|
if explicit:
|
|
return explicit, "--session"
|
|
for name in (SESSION_ENV, HARNESS_SESSION_ENV):
|
|
if os.environ.get(name):
|
|
return os.environ[name], name
|
|
return None, None
|
|
|
|
|
|
def trace_root(root: Path) -> Path:
|
|
override = os.environ.get("WIKI_TRACE_DIR")
|
|
return Path(override) if override else root / "reports" / "telemetry"
|
|
|
|
|
|
def collect_trace(root: Path, explicit):
|
|
"""The caller's session trace, or a listing of what exists instead."""
|
|
session, source = resolve_trace_session(explicit)
|
|
base = trace_root(root)
|
|
if session:
|
|
path = base / session_slug(session) / "trace.jsonl"
|
|
if path.is_file():
|
|
return {"found": True, "session": session, "source": source,
|
|
"text": read_text(path)}
|
|
listing = []
|
|
if base.is_dir():
|
|
for child in sorted(base.iterdir()):
|
|
if child.is_dir() and not child.name.startswith("bugreport-"):
|
|
trace = child / "trace.jsonl"
|
|
listing.append({"session_directory": child.name,
|
|
"bytes": trace.stat().st_size if trace.is_file() else 0})
|
|
return {"found": False, "session": session, "source": source, "listing": listing}
|
|
|
|
|
|
WIKITOOL_CALLS = (
|
|
("version show", ("version", "show"), "inherited"),
|
|
("doctor", ("doctor",), "inherited"),
|
|
("budget status", ("budget", "status"), "inherited"),
|
|
("instructions verify", ("instructions", "verify"), "own"),
|
|
("docs verify", ("docs", "verify"), "own"),
|
|
)
|
|
|
|
|
|
def run_wikitool(root: Path, launcher, own_session: str) -> dict:
|
|
"""Five fixed read commands. `version show` is the startup probe: if it fails,
|
|
`wikitool` counts as not started and nothing else is run."""
|
|
outputs = []
|
|
started = False
|
|
reason = None
|
|
for label, args, session in WIKITOOL_CALLS:
|
|
env = dict(os.environ)
|
|
if session == "own":
|
|
env[SESSION_ENV] = own_session
|
|
result = run(launcher + list(args), root, env=env, timeout=TIMEOUT_SECONDS)
|
|
prefix = "%s=%s " % (SESSION_ENV, own_session) if session == "own" else ""
|
|
outputs.append({
|
|
"label": label,
|
|
"file": "wikitool/%s.txt" % label.replace(" ", "-"),
|
|
"session": ("%s (set by the collector)" % own_session) if session == "own"
|
|
else "inherited from the caller",
|
|
"text": (
|
|
"command: %s%s\nsession: %s\nexit: %s\n%s\n--- stdout ---\n%s\n--- stderr ---\n%s\n"
|
|
% (prefix, " ".join(launcher + list(args)),
|
|
"own" if session == "own" else "inherited",
|
|
result.returncode if result.returncode is not None else result.note,
|
|
("note: " + result.note) if result.note and result.returncode is not None
|
|
else "", result.stdout, result.stderr)
|
|
),
|
|
})
|
|
if label == "version show":
|
|
if not result.ok:
|
|
reason = "version show %s: %s" % (
|
|
("exit %s" % result.returncode) if result.returncode is not None
|
|
else result.note, result.first_line())
|
|
break
|
|
started = True
|
|
return {"started": started, "reason": reason, "outputs": outputs}
|
|
|
|
|
|
# ------------------------------------------------------------------------------ bundle
|
|
|
|
|
|
class Bundle:
|
|
def __init__(self, directory: Path) -> None:
|
|
self.directory = directory
|
|
self.files = []
|
|
|
|
def add(self, rel: str, text: str, layer: str, description: str, note: str = "") -> None:
|
|
write_text(self.directory / rel, text)
|
|
self.files.append((rel, layer, description, note))
|
|
|
|
|
|
def stamp_now() -> str:
|
|
return datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
|
|
|
|
|
|
def unique_stamp(out: Path) -> str:
|
|
stamp = stamp_now()
|
|
candidate, n = stamp, 1
|
|
while (out / ("bugreport-" + candidate)).exists() or (out / ("bugreport-%s.zip" % candidate)).exists():
|
|
n += 1
|
|
candidate = "%s-%d" % (stamp, n)
|
|
return candidate
|
|
|
|
|
|
def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str:
|
|
lines = ["# Bug report bundle", "",
|
|
"Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"),
|
|
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", PRIVACY_NOTICE, "",
|
|
"## Options", "",
|
|
"- chronology: %s" % ("supplied" if args.chronology else "not supplied"),
|
|
"- transcripts: %d" % len(args.transcript),
|
|
"- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"),
|
|
"- session trace: %s" % (
|
|
"excluded (`--no-trace`)" if args.no_trace else
|
|
("included" if trace and trace.get("found") else "not found")),
|
|
"", "## wikitool", ""]
|
|
if wikitool is None:
|
|
lines.append("`wikitool` did not start: no launcher found under `tools/`.")
|
|
elif wikitool["started"]:
|
|
lines.append("`wikitool` started. Five read commands ran; outputs are verbatim.")
|
|
else:
|
|
lines.append("`wikitool` did not start (%s). The four other commands were not run."
|
|
% wikitool["reason"])
|
|
if wikitool:
|
|
lines += ["", "| Command | Session id |", "|---|---|"]
|
|
for out in wikitool["outputs"]:
|
|
lines.append("| `%s` | %s |" % (out["label"], out["session"]))
|
|
caller, source = resolve_trace_session(args.session)
|
|
lines += ["", "Caller session id: %s" % (
|
|
"%s (from %s)" % (caller, source) if caller else
|
|
"not set; `wikitool` falls back to its parent process id"), ""]
|
|
if trace and not trace.get("found") and not args.no_trace:
|
|
lines += ["No trace found for the caller's session%s. Existing session directories "
|
|
"(name and size only):" % (
|
|
" `%s`" % trace["session"] if trace.get("session") else ""), ""]
|
|
for item in trace.get("listing", [])[:100]:
|
|
lines.append("- `%s` (%d bytes)" % (item["session_directory"], item["bytes"]))
|
|
if not trace.get("listing"):
|
|
lines.append("- none")
|
|
lines.append("")
|
|
lines += ["## Files", "", "| File | Layer | Contents | Note |", "|---|---|---|---|"]
|
|
for rel, layer, description, note in sorted(bundle.files, key=lambda f: f[0]):
|
|
lines.append("| `%s` | %s | %s | %s |" % (rel, layer, description, note))
|
|
lines.append("| `MANIFEST.md` | - | this file | |")
|
|
if gaps:
|
|
lines += ["", "## Gaps", ""] + ["- %s" % gap for gap in gaps]
|
|
return "\n".join(lines) + "\n"
|
|
|
|
|
|
def final_pass(bundle_dir: Path, red: Redactor, shield, skip) -> None:
|
|
for path in sorted(bundle_dir.rglob("*")):
|
|
if not path.is_file():
|
|
continue
|
|
rel = path.relative_to(bundle_dir).as_posix()
|
|
text = read_text(path)
|
|
cleaned = red.scrub(text)
|
|
if shield is not None and rel not in skip:
|
|
cleaned = shield.apply(cleaned)
|
|
if cleaned != text:
|
|
write_text(path, cleaned)
|
|
|
|
|
|
def make_zip(bundle_dir: Path, archive: Path) -> None:
|
|
with zipfile.ZipFile(str(archive), "w", zipfile.ZIP_DEFLATED) as zf:
|
|
for path in sorted(bundle_dir.rglob("*")):
|
|
if path.is_file():
|
|
zf.write(str(path), bundle_dir.name + "/" + path.relative_to(bundle_dir).as_posix())
|
|
|
|
|
|
def parse_args(argv):
|
|
parser = argparse.ArgumentParser(
|
|
prog="bugreport.py",
|
|
description="Collect a bug-report bundle. Removes secrets, keeps page titles out "
|
|
"unless --titles is given, uploads nothing.",
|
|
)
|
|
parser.add_argument("--chronology", help="the agent's fact-only chronology (Markdown)")
|
|
parser.add_argument("--transcript", action="append", default=[],
|
|
help="a harness transcript to include (repeatable; only on request)")
|
|
group = parser.add_mutually_exclusive_group()
|
|
group.add_argument("--session", help="session id whose trace to include")
|
|
group.add_argument("--no-trace", action="store_true", help="leave the session trace out")
|
|
parser.add_argument("--titles", action="store_true",
|
|
help="keep page titles and paths (writes tree-paths.txt)")
|
|
parser.add_argument("--root", help="checkout root (default: parent of tools/)")
|
|
parser.add_argument("--out", help="output directory (default: <root>/reports)")
|
|
return parser.parse_args(argv)
|
|
|
|
|
|
def main(argv=None) -> int:
|
|
args = parse_args(argv)
|
|
root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent
|
|
out = Path(args.out).resolve() if args.out else root / "reports"
|
|
|
|
inputs = [("chronology", args.chronology)] + [("transcript", t) for t in args.transcript]
|
|
for label, value in inputs:
|
|
if value and not Path(value).is_file():
|
|
sys.stderr.write("bugreport: %s file not found: %s\n" % (label, value))
|
|
return 1
|
|
|
|
try:
|
|
out.mkdir(parents=True, exist_ok=True)
|
|
stamp = unique_stamp(out)
|
|
bundle_dir = out / ("bugreport-" + stamp)
|
|
bundle_dir.mkdir()
|
|
return _collect(args, root, out, stamp, bundle_dir)
|
|
except OSError as exc:
|
|
sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc)
|
|
return 1
|
|
|
|
|
|
def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int:
|
|
red = Redactor()
|
|
bundle = Bundle(bundle_dir)
|
|
gaps = []
|
|
|
|
trace = None
|
|
if not args.no_trace:
|
|
trace = guarded(collect_trace, root, args.session)
|
|
if trace.get("found"):
|
|
bundle.add("trace.jsonl", trace["text"], "3", "session trace of the caller",
|
|
MAY_CONTAIN_CONTENT)
|
|
|
|
bundle.add("environment.json", dump_json(guarded(collect_environment, root, red)), "1",
|
|
"OS, Python, shells, harness, PATH, environment variables (names; values for a "
|
|
"fixed list), git configuration, venv, line endings, Windows details")
|
|
|
|
stack = guarded(collect_stack, root, red)
|
|
bundle.add("stack.json", dump_json(stack), "2",
|
|
"VERSION, `.wikitool-*.json` (secrets removed), git status and log")
|
|
|
|
content_paths = []
|
|
tree = {}
|
|
for name in CONTENT_DIRS:
|
|
summary, files = _safe_scan(root, name)
|
|
tree[name] = summary
|
|
content_paths.extend(files)
|
|
bundle.add("tree-structure.json", dump_json(tree), "2",
|
|
"kb/ and raw/: counts, depth, path lengths, name problems - no content")
|
|
if args.titles:
|
|
bundle.add("tree-paths.txt", "\n".join(sorted(content_paths)) + "\n", "2",
|
|
"all relative paths under kb/ and raw/", "contains page titles")
|
|
|
|
wikitool = None
|
|
launcher = launcher_command(root)
|
|
if launcher:
|
|
wikitool = run_wikitool(root, launcher, "bugreport-" + stamp)
|
|
for item in wikitool["outputs"]:
|
|
bundle.add(item["file"], item["text"], "3", "verbatim output of `%s`" % item["label"])
|
|
else:
|
|
gaps.append("no launcher found under tools/, so no wikitool command ran")
|
|
|
|
if args.chronology:
|
|
bundle.add("CHRONOLOGY.md", read_text(Path(args.chronology)), "4",
|
|
"the agent's chronology", MAY_CONTAIN_CONTENT)
|
|
else:
|
|
gaps.append("no chronology was supplied")
|
|
used = set()
|
|
for source in args.transcript:
|
|
name = Path(source).name
|
|
target, n = name, 1
|
|
while target in used:
|
|
n += 1
|
|
target = "%d-%s" % (n, name)
|
|
used.add(target)
|
|
bundle.add("transcripts/" + target, read_text(Path(source)), "4",
|
|
"harness transcript", MAY_CONTAIN_CONTENT)
|
|
|
|
write_text(bundle_dir / "MANIFEST.md",
|
|
build_manifest(args, root, stamp, bundle, wikitool, trace, gaps))
|
|
|
|
shield = None
|
|
if not args.titles:
|
|
shield = TitleShield(
|
|
rel for rel in content_paths if rel.rsplit("/", 1)[-1] not in STACK_NAMES
|
|
)
|
|
final_pass(bundle_dir, red, shield, {"tree-paths.txt"})
|
|
|
|
archive = out / ("bugreport-%s.zip" % stamp)
|
|
make_zip(bundle_dir, archive)
|
|
|
|
print("Bundle: %s" % bundle_dir)
|
|
print("Archive: %s" % archive)
|
|
print()
|
|
print(PRIVACY_NOTICE)
|
|
return 0
|
|
|
|
|
|
def _safe_scan(root: Path, name: str):
|
|
try:
|
|
return scan_tree(root, name)
|
|
except Exception as exc: # noqa: BLE001
|
|
return {"collector_error": "%s: %s" % (type(exc).__name__, exc)}, []
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|