#!/usr/bin/env python3 """Collect a bug-report bundle from this checkout. Runs on the base Python with the standard library only, and imports nothing from `chemenu`: the case it exists for is a checkout where `wikitool` does not start - no venv, a broken package, a Python that is too old. Syntax stays at Python 3.8 so that even an old interpreter can still produce a report. python tools/bugreport.py [--chronology FILE] [--transcript FILE]... [--session ID | --no-trace] [--titles] [--pseudonymise] [--root DIR] [--out DIR] python tools/bugreport.py --bundle DIR --candidates FILE The bundle is `/bugreport-/` plus a zip beside it, in four layers: the environment, the stack, what `wikitool` prints (if it starts), and the agent's chronology. Secrets are always removed. Page titles are kept out of everything this script generates unless `--titles` is given. With `--pseudonymise` known identities are replaced by consistent, shape-preserving placeholders (stage 1); `--bundle`/`--candidates` applies the names a model found beyond them (stage 2). The mapping stays beside the bundle, never in it. Nothing is uploaded. Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script opens no gate. See instructions/bug-report.md. """ from __future__ import annotations import argparse import datetime import getpass import hashlib import hmac import json import os import platform import re import secrets import shutil import socket import subprocess import sys import unicodedata import zipfile from pathlib import Path TIMEOUT_SECONDS = 300 PROBE_TIMEOUT_SECONDS = 20 REMOVED = "" NOT_COLLECTED = "" SESSION_ENV = "WIKITOOL_SESSION_ID" HARNESS_SESSION_ENV = "CLAUDE_CODE_SESSION_ID" SECRET_NAME = re.compile(r"token|password|secret|key|auth", re.IGNORECASE) HARNESS_MARKERS = ( (HARNESS_SESSION_ENV, "claude-code"), ) ENV_ALLOW_EXACT = frozenset( { "PATH", "PATHEXT", "SHELL", "TERM", "TERM_PROGRAM", "TERM_PROGRAM_VERSION", "COLORTERM", "LANG", "LANGUAGE", "VIRTUAL_ENV", "MSYSTEM", "COMSPEC", "PSMODULEPATH", } | {name for name, _ in HARNESS_MARKERS} ) ENV_ALLOW_PREFIX = ("LC_", "PYTHON", "WIKI_", "WIKITOOL_", "CHEMENU_") # Names under kb/ that the stack generates or owns: not titles, and unreadable # in a bundle if they are masked. STACK_NAMES = frozenset( { "CONTRACT.md", "CONVENTIONS.md", "CONVENTIONS.md.template", "COLLECTION.md", "COLLECTION.md.template", "INDEX.md", "index.md", "log.md", "provenance.md", ".gitkeep", } ) CONTENT_DIRS = ("kb", "raw") CONTENT_COMMIT_DIRS = ("kb", "raw", "work") MAY_CONTAIN_CONTENT = "may contain page content and titles" PRIVACY_NOTICE = ( "This bundle is not pseudonymised. It contains private data: machine, user and path " "names, PATH entries, git remotes and commit subjects and - if included - the session " "trace, the chronology and transcripts, which may contain page content and titles. " "Secrets are removed, but read the bundle before you share it. Choose the channel " "yourself; neither this tool nor the agent uploads anything." ) STAGE1_NOTICE = ( "Pseudonymisation: stage 1 applied, stage 2 not yet applied. Identities this script could read " "from the machine (user, host, home and repository paths, git identity, remotes) are replaced by " "consistent placeholders that keep length, spaces, hyphens, character classes, separators and depth. " "Names it cannot know - people, companies, customers, internal hosts, projects - may remain, above " "all in the trace, the chronology and transcripts. Secrets are removed. Read the bundle before you " "share it. Choose the channel yourself; neither this tool nor the agent uploads anything." ) RESIDUAL_NOTICE = ( "Pseudonymisation: stage 1 and stage 2 applied. Stage 2 is the judgement of a model: it can miss " "names, companies, hosts and projects, above all in the free text of a trace or transcript over " "100 KB, which the model did not read in full. The forms - lengths, character classes, separators " "and depth - are kept on purpose, so a rare name can still be recognisable by its shape. Read the " "bundle before you share it. Choose the channel yourself; neither this tool nor the agent uploads " "anything." ) PUBLIC_ORIGIN = "gitea.nehmer.net/torben/chemenu" # same origin as chemenu/version.py; this script imports nothing from chemenu MIN_IDENTITY = 3 NON_ASCII_LOWER = "äöüéèêàáâçñõøåæ" WORD = re.compile(r"[^\W_]+") VOCABULARY = frozenset( w.lower() for w in ( "Windows System32 SysWOW64 Program Files ProgramData Users Public AppData Local LocalLow " "Roaming Temp Microsoft WindowsApps Programs Documents Desktop Downloads OneDrive " "DESKTOP LAPTOP " "home usr local bin opt etc var tmp mnt src Library Applications " "Python Git Scripts venv chocolatey " "chemenu wikitool tools reports kb raw bugreport " "GmbH AG KG SE Inc Ltd LLC " "removed collected title depth len space " "json jsonl md txt py ps1 zip cfg template" ).split() ) WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE) WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]') WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]") URL_USERINFO = re.compile(r"(?P[A-Za-z][A-Za-z0-9+.\-]*)://(?P[^/\s@\"'<>]+)@") AUTH_HEADER = re.compile(r"(?im)(authorization\s*[:=]\s*)(?:(?:bearer|basic|token)\s+)?\S+") BEARER = re.compile(r"(?i)\b(bearer)\s+[A-Za-z0-9._~+/=\-]{8,}") # --------------------------------------------------------------------------- secrets class Redactor: """Every secret value found while collecting, replaced again in a final pass.""" def __init__(self) -> None: self.values = set() def add(self, value) -> None: if isinstance(value, str) and len(value) >= 4: self.values.add(value) def add_any(self, node) -> None: if isinstance(node, dict): for item in node.values(): self.add_any(item) elif isinstance(node, (list, tuple)): for item in node: self.add_any(item) else: self.add(node) def scrub_urls(self, text: str) -> str: def repl(match): scheme, userinfo = match.group("scheme"), match.group("userinfo") if scheme.lower() in ("http", "https"): self.add(userinfo) for part in userinfo.split(":"): self.add(part) return scheme + "://" if ":" in userinfo: user, password = userinfo.split(":", 1) self.add(password) return scheme + "://" + user + "@" return match.group(0) return URL_USERINFO.sub(repl, text) def scrub(self, text: str) -> str: text = self.scrub_urls(text) text = AUTH_HEADER.sub(lambda m: m.group(1) + REMOVED, text) text = BEARER.sub(lambda m: m.group(1) + " " + REMOVED, text) forms = set() for value in self.values: forms.add(value) escaped = json.dumps(value, ensure_ascii=False)[1:-1] if escaped != value: forms.add(escaped) for form in sorted(forms, key=len, reverse=True): text = text.replace(form, REMOVED) return text def clean_json(self, node): if isinstance(node, dict): out = {} for key, value in node.items(): if SECRET_NAME.search(str(key)): self.add_any(value) out[key] = REMOVED else: out[key] = self.clean_json(value) return out if isinstance(node, list): return [self.clean_json(item) for item in node] if isinstance(node, str): return self.scrub_urls(node) return node # ------------------------------------------------------------------------ title shield def _shape(rel: str) -> str: parts = rel.split("/") name = parts[-1] stem, dot, suffix = name.rpartition(".") flags = ["depth=%d" % len(parts), "len=%d" % len(rel)] if " " in rel: flags.append("space") if any(ord(ch) > 127 for ch in rel): flags.append("non-ascii") return "%s/<%s>%s" % (parts[0], " ".join(flags), (dot + suffix) if dot and stem else "") class TitleShield: """Replaces the relative path of a file under kb/ or raw/ with its shape. A title that stands as bare prose is not found, on purpose: replacing titles word by word would hit a title like "Git" everywhere in the bundle. """ def __init__(self, rel_paths) -> None: self.forms = {} for rel in rel_paths: shape = _shape(rel) for form in (rel, rel.replace("/", "\\"), rel.replace("/", "\\\\")): self.forms[form] = shape self.lengths = sorted({len(form) for form in self.forms}, reverse=True) self.start = re.compile(r"(?:%s)(?:/|\\)" % "|".join(CONTENT_DIRS)) def apply(self, text: str) -> str: if self.forms: out, last = [], 0 for match in self.start.finditer(text): i = match.start() if i < last: continue for length in self.lengths: shape = self.forms.get(text[i:i + length]) if shape: out.append(text[last:i]) out.append(shape) last = i + length break out.append(text[last:]) text = "".join(out) return WIKILINK.sub(lambda m: "[[]]" % _title_flags(m.group(1)), text) def _title_flags(title: str) -> str: flags = ["len=%d" % len(title)] if " " in title: flags.append("space") if any(ord(ch) > 127 for ch in title): flags.append("non-ascii") return " ".join(flags) # ------------------------------------------------------------------------ pseudonymisation def _words(text: str): return [m.group(0).lower() for m in WORD.finditer(text)] class Pseudonymiser: """Consistent, shape-preserving replacement of identities, word by word. A placeholder hangs on the word, not on the identity: "Beispiel" is replaced the same way inside "OneDrive - Beispiel GmbH" (stage 1) and inside the candidate "Beispiel GmbH" (stage 2), so the two stages agree without coordinating. """ def __init__(self, salt=None) -> None: self.salt = salt or secrets.token_hex(16) self.words = {} self.keep = set() self.entries = [] self.skipped = 0 self.stage = 1 self._index = {} self._taken = set(VOCABULARY) self._regex = {} self._forms = {} # -- persistence def to_json(self, bundle_name: str) -> dict: return { "schema": 1, "bundle": bundle_name, "stage": self.stage, "salt": self.salt, "keep": sorted(self.keep), "skipped": self.skipped, "words": self.words, "entries": self.entries, } @classmethod def from_json(cls, data: dict) -> "Pseudonymiser": pz = cls(data["salt"]) pz.stage = data.get("stage", 1) pz.keep = set(data.get("keep", [])) pz.skipped = data.get("skipped", 0) pz.words = dict(data.get("words", {})) pz.entries = list(data.get("entries", [])) for key, pseudo in pz.words.items(): pz._taken.add(key) pz._taken.add(pseudo) for index, entry in enumerate(pz.entries): pz._index[entry["original"].lower()] = index pz._taken.update(_words(entry["original"])) return pz # -- placeholders def _stream(self, key: str, counter: int): block = 0 salt = bytes.fromhex(self.salt) while True: message = "%s\x00%d\x00%d" % (key, counter, block) for byte in hmac.new(salt, message.encode("utf-8"), hashlib.sha256).digest(): yield byte block += 1 def pseudo_word(self, word: str) -> str: key = word.lower() if key in VOCABULARY or key in self.keep: return word base = self.words.get(key) if base is None: counter = 0 while True: stream = self._stream(key, counter) chars = [] for char in word: byte = next(stream) if char.isdigit(): chars.append(str(byte % 10)) elif char.isascii(): chars.append(chr(ord("a") + byte % 26)) else: chars.append(NON_ASCII_LOWER[byte % len(NON_ASCII_LOWER)]) base = "".join(chars) if (base != key and base not in self._taken) or counter >= 500: break counter += 1 self.words[key] = base self._taken.add(base) if len(base) != len(word): base = (base * (len(word) // max(len(base), 1) + 1))[:len(word)] return "".join( b.upper() if orig.isupper() and not b.isdigit() else b for orig, b in zip(word, base) ) def pseudo_text(self, text: str) -> str: return WORD.sub(lambda m: self.pseudo_word(m.group(0)), text) # -- identities def add_identity(self, original: str, kind: str, stage: int = 1) -> bool: original = original.strip() if not original or original.lower() in self._index: return False words = _words(original) if len(original) < MIN_IDENTITY or not words: self.skipped += 1 return False if kind in ("host", "domain", "remote") and "." in original: label = _words(original)[-1] if not label.isdigit(): self.keep.add(label) if all(w in VOCABULARY or w in self.keep for w in words): self.skipped += 1 return False self._taken.update(words) self._index[original.lower()] = len(self.entries) self.entries.append({"original": original, "placeholder": None, "kind": kind, "stage": stage, "count": 0}) self._regex = {} return True def finish(self) -> None: for entry in self.entries: if entry["placeholder"] is None: entry["placeholder"] = self.pseudo_text(entry["original"]) # -- applying def _build(self, stage) -> None: self.finish() forms = {PUBLIC_ORIGIN.lower(): (None, False)} for entry in self.entries: if stage is not None and entry["stage"] != stage: continue raw = entry["original"] forms.setdefault(raw.lower(), (entry, False)) escaped = json.dumps(raw, ensure_ascii=True)[1:-1] if escaped.lower() != raw.lower(): forms.setdefault(escaped.lower(), (entry, True)) names = sorted(forms, key=len, reverse=True) self._regex[stage] = (forms, re.compile( r"(?<![^\W_])(?:%s)(?![^\W_])" % "|".join(re.escape(n) for n in names), re.IGNORECASE )) def apply(self, text: str, stage=None) -> str: if not any(stage is None or e["stage"] == stage for e in self.entries): return text if stage not in self._regex: self._build(stage) forms, regex = self._regex[stage] def repl(match): found = forms.get(match.group(0).lower()) if found is None or found[0] is None: return match.group(0) entry, escaped = found if escaped: try: decoded = json.loads('"%s"' % match.group(0)) except ValueError: return match.group(0) entry["count"] += 1 return json.dumps(self.pseudo_text(decoded), ensure_ascii=True)[1:-1] entry["count"] += 1 return self.pseudo_text(match.group(0)) return regex.sub(repl, text) def find(self, text: str, original: str) -> int: raw = re.escape(original) escaped = re.escape(json.dumps(original, ensure_ascii=True)[1:-1]) pattern = re.compile(r"(?<![^\W_])(?:%s|%s)(?![^\W_])" % (raw, escaped), re.IGNORECASE) return len(pattern.findall(text)) def placeholder_words(self) -> set: return set(self.words.values()) def _path_parts(value: str): return [part for part in re.split(r"[\\/]+", value) if part] def _remote_identities(url: str): """(original, kind) pairs for one remote URL. The stack's public origin is exempt.""" url = url.strip() if not url or PUBLIC_ORIGIN in url.lower().replace("\\", "/"): return [] found = [] scheme = re.match(r"^[A-Za-z][A-Za-z0-9+.\-]*://", url) scp = re.match(r"^(?:[^@/\\:\s]+@)?([^/\\:\s]{2,}):(?![/\\])(.*)$", url) if scheme: rest = url[scheme.end():] netloc, _, path = rest.partition("/") host = netloc.rsplit("@", 1)[-1].split(":")[0] found.append((host, "remote")) parts = _path_parts(path) elif scp: found.append((scp.group(1), "remote")) parts = _path_parts(scp.group(2)) else: parts = _path_parts(url) found.extend((part, "remote") for part in parts) return found def collect_identities(root: Path): """Identities readable from this machine, in the order they are registered.""" found = [] user = set() try: user.add(getpass.getuser()) except Exception: # noqa: BLE001 - no account name is not a failure pass for name in ("USER", "USERNAME", "LOGNAME"): if os.environ.get(name): user.add(os.environ[name]) found.extend((value, "user name") for value in sorted(user)) if os.environ.get("USERDOMAIN"): found.append((os.environ["USERDOMAIN"], "domain")) if os.environ.get("COMPUTERNAME"): found.append((os.environ["COMPUTERNAME"], "host")) try: found.append((socket.gethostname(), "host")) except OSError: pass found.extend((part, "home path") for part in _path_parts(os.path.expanduser("~"))) found.extend((part, "repo path") for part in _path_parts(str(root))) for key in ("user.name", "user.email"): result = run(["git", "config", "--get", key], root) value = result.stdout.strip() if result.ok else "" if value: found.append((value, "git identity")) if key == "user.email" and "@" in value: local, _, domain = value.rpartition("@") found.append((local, "git identity")) found.append((domain, "domain")) remotes = run(["git", "remote", "-v"], root) if remotes.ok: for line in remotes.stdout.splitlines(): fields = line.split("\t", 1) if len(fields) == 2: found.extend(_remote_identities(re.sub(r"\s+\((?:fetch|push)\)\s*$", "", fields[1]))) return found def text_files(bundle_dir: Path): return [p for p in sorted(bundle_dir.rglob("*")) if p.is_file()] def pseudonymise_bundle(bundle_dir: Path, pz: Pseudonymiser, stage=None) -> None: for path in text_files(bundle_dir): text = read_text(path) changed = pz.apply(text, stage) if changed != text: write_text(path, changed) EMAIL_RE = re.compile(r"[\w.+\-]+@[\w\-]+(?:\.[\w\-]+)+") URL_RE = re.compile(r"[A-Za-z][A-Za-z0-9+.\-]*://[^\s\"'<>)\]]+") PATH_RE = re.compile(r"(?:[A-Za-z]:)?(?:[\\/]+[^\\/\s\"'<>|:*?,;()\[\]{}]+){2,}") def build_review(bundle_dir: Path, pz: Pseudonymiser) -> str: """What stage 1 leaves structured and unmasked, for the model that does stage 2.""" found = {} def note(kind: str, value: str, rel: str) -> None: words = _words(value) if len(value) < MIN_IDENTITY or not words: return if all(w in VOCABULARY or w in pz.keep or w in pz.placeholder_words() for w in words): return entry = found.setdefault((kind, value), {"count": 0, "files": []}) entry["count"] += 1 if rel not in entry["files"]: entry["files"].append(rel) for path in text_files(bundle_dir): rel = path.relative_to(bundle_dir).as_posix() text = read_text(path) for match in EMAIL_RE.finditer(text): note("mail address", match.group(0), rel) for match in URL_RE.finditer(text): rest = match.group(0).split("://", 1)[1] netloc, _, tail = rest.partition("/") note("host", netloc.rsplit("@", 1)[-1], rel) for part in _path_parts(tail): note("url segment", part, rel) for match in PATH_RE.finditer(text): for part in _path_parts(match.group(0)): note("path component", part, rel) lines = [ "# Review list for stage 2 - local only, never part of the bundle. It holds what stage 1", "# left in the bundle, which can still be a real name. One line: kind, count, value, files.", ] for (kind, value), info in sorted(found.items(), key=lambda kv: (-kv[1]["count"], kv[0])): lines.append("%s\t%d\t%s\t%s" % (kind, info["count"], value, ", ".join(info["files"][:5]))) return "\n".join(lines) + "\n" def pseudonym_table(pz: Pseudonymiser) -> str: lines = ["| Placeholder | Kind | Stage | Occurrences |", "|---|---|---|---|"] for entry in pz.entries: placeholder = entry["placeholder"].replace("|", "\\|") lines.append("| `%s` | %s | %d | %d |" % ( placeholder, entry["kind"], entry["stage"], entry["count"])) lines += ["", "%d identities were left unchanged (under %d characters, or system vocabulary); " "their values are not listed." % (pz.skipped, MIN_IDENTITY)] return "\n".join(lines) def replace_section(manifest: str, heading: str, body: str) -> str: sections = re.split(r"(?m)^(?=## )", manifest) block = "## %s\n\n%s\n" % (heading, body) for index, section in enumerate(sections): if section.startswith("## %s\n" % heading): sections[index] = block break else: sections.append(block) return "\n\n".join(part.strip("\n") for part in sections if part.strip()) + "\n" # ---------------------------------------------------------------------------- helpers class Result: def __init__(self, returncode, stdout, stderr, note=None) -> None: self.returncode = returncode self.stdout = stdout self.stderr = stderr self.note = note @property def ok(self) -> bool: return self.returncode == 0 def first_line(self) -> str: for text in (self.stderr, self.stdout, self.note or ""): for line in text.splitlines(): if line.strip(): return line.strip() return "" def _decode(data) -> str: if data is None: return "" if isinstance(data, str): return data return data.decode("utf-8", errors="replace") def run(cmd, cwd, env=None, timeout=PROBE_TIMEOUT_SECONDS) -> Result: try: proc = subprocess.run( cmd, cwd=str(cwd), env=env, stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout, ) except subprocess.TimeoutExpired as exc: return Result(None, _decode(exc.stdout), _decode(exc.stderr), "timed out after %d s" % timeout) except OSError as exc: return Result(None, "", "", "could not start: %s" % exc) return Result(proc.returncode, _decode(proc.stdout), _decode(proc.stderr)) def read_text(path: Path) -> str: with open(str(path), "r", encoding="utf-8", errors="replace", newline="") as handle: return handle.read() def write_text(path: Path, text: str) -> None: path.parent.mkdir(parents=True, exist_ok=True) with open(str(path), "w", encoding="utf-8", newline="\n") as handle: handle.write(text) def dump_json(data) -> str: return json.dumps(data, indent=2, ensure_ascii=False) + "\n" def guarded(fn, *args): """A broken layer must not take the whole report down.""" try: return fn(*args) except Exception as exc: # noqa: BLE001 - the report is the point return {"collector_error": "%s: %s" % (type(exc).__name__, exc)} def session_slug(value: str) -> str: return re.sub(r"[^A-Za-z0-9._-]+", "__", value).strip("_") or "unknown" # ------------------------------------------------------------------- layer 1: environment def _version_line(cmd, cwd) -> str: result = run(cmd, cwd) return result.first_line() if result.ok or result.first_line() else (result.note or "") def collect_environment(root: Path, red: Redactor) -> dict: data = { "os": { "platform": platform.platform(), "system": platform.system(), "release": platform.release(), "version": platform.version(), "machine": platform.machine(), }, "python": _collect_python(root), "shells": _collect_shells(root), "harness": _collect_harness(), "path_entries": [ {"entry": entry, "exists": os.path.isdir(entry)} for entry in os.environ.get("PATH", "").split(os.pathsep) if entry ], "env": _collect_env(red), "tools_config": _collect_tools_config(root, red), "git": _collect_git_config(root, red), "venv": _collect_venv(root, red), "line_endings": _collect_line_endings(root), "gitattributes": _read_optional(root / ".gitattributes", red), "windows": _collect_windows(root), } return data def _collect_python(root: Path) -> dict: running = { "executable": sys.executable, "version": sys.version.replace("\n", " "), "version_info": list(sys.version_info[:3]), } candidates = [] for name, extra in (("python3", []), ("python", []), ("py", ["-3"])): path = shutil.which(name) entry = {"name": " ".join([name] + extra), "path": path} if path: entry["version"] = _version_line([path] + extra + ["--version"], root) entry["windows_store_alias"] = "windowsapps" in path.lower() candidates.append(entry) return {"running": running, "candidates": candidates} def _collect_shells(root: Path) -> dict: shells = {"SHELL": os.environ.get("SHELL"), "COMSPEC": os.environ.get("COMSPEC")} for name in ("bash", "zsh", "fish"): path = shutil.which(name) shells[name] = {"path": path, "version": _version_line([path, "--version"], root) if path else None} for name in ("pwsh", "powershell"): path = shutil.which(name) shells[name] = { "path": path, "version": _version_line( [path, "-NoProfile", "-Command", "$PSVersionTable.PSVersion.ToString()"], root ) if path else None, } return shells def _collect_harness() -> dict: detected = [ {"marker": name, "harness": harness} for name, harness in HARNESS_MARKERS if os.environ.get(name) ] if os.environ.get("TERM_PROGRAM") == "vscode": detected.append({"marker": "TERM_PROGRAM=vscode", "harness": "vscode terminal"}) return {"detected": detected or "unrecognised"} def _env_allowed(name: str) -> bool: upper = name.upper() return upper in ENV_ALLOW_EXACT or upper.startswith(ENV_ALLOW_PREFIX) def _collect_env(red: Redactor) -> dict: env = {} for name in sorted(os.environ): value = os.environ[name] if SECRET_NAME.search(name): red.add(value) env[name] = REMOVED if _env_allowed(name) else NOT_COLLECTED elif _env_allowed(name): env[name] = value else: env[name] = NOT_COLLECTED return env def _collect_tools_config(root: Path, red: Redactor) -> dict: path = root / ".wikitool-tools.json" if not path.is_file(): return {"status": "absent"} try: parsed = json.loads(read_text(path)) except ValueError as exc: return {"status": "unparseable", "error": str(exc)} checks = [] def walk(node, where): if isinstance(node, dict): for key, value in node.items(): walk(value, where + [str(key)]) elif isinstance(node, list): for index, value in enumerate(node): walk(value, where + [str(index)]) elif isinstance(node, str) and os.path.isabs(node): checks.append({"at": ".".join(where), "path": node, "exists": os.path.exists(node)}) walk(parsed, []) return {"status": "present", "content": red.clean_json(parsed), "path_checks": checks} def _collect_git_config(root: Path, red: Redactor) -> dict: version = run(["git", "--version"], root) if not version.ok: return {"available": False, "error": version.note or version.first_line()} config = {} for key in ("core.autocrlf", "core.longpaths", "core.filemode"): result = run(["git", "config", "--get", key], root) config[key] = result.stdout.strip() if result.ok else "(unset)" name = run(["git", "config", "--get", "user.name"], root) config["user.name is set"] = bool(name.ok and name.stdout.strip()) remotes = run(["git", "remote", "-v"], root) return { "available": True, "version": version.stdout.strip(), "config": config, "remotes": red.scrub_urls(remotes.stdout).splitlines() if remotes.ok else ["(not a repository, or git failed: %s)" % remotes.first_line()], } def _collect_venv(root: Path, red: Redactor) -> dict: venv = root / "tools" / ".venv" if not venv.is_dir(): return {"present": False} layout = ("bin" if (venv / "bin").is_dir() else None) or ( "Scripts" if (venv / "Scripts").is_dir() else None) cfg = venv / "pyvenv.cfg" return { "present": True, "layout": layout or "unknown", "python_found": any( (venv / sub / exe).exists() for sub, exe in (("bin", "python"), ("Scripts", "python.exe")) ), "pyvenv.cfg": read_text(cfg) if cfg.is_file() else None, } def _collect_line_endings(root: Path) -> dict: tools = root / "tools" files = [tools / "wikitool"] + (sorted(tools.glob("*.ps1")) if tools.is_dir() else []) result = {} for path in files: if not path.is_file(): continue raw = path.read_bytes() crlf = raw.count(b"\r\n") result[path.name] = {"crlf": crlf, "lf_only": raw.count(b"\n") - crlf} return result def _read_optional(path: Path, red: Redactor): return red.scrub_urls(read_text(path)) if path.is_file() else None def _collect_windows(root: Path) -> dict: if sys.platform != "win32": return {"applicable": False} data = {"applicable": True} try: import winreg # noqa: PLC0415 - only exists on Windows with winreg.OpenKey( winreg.HKEY_LOCAL_MACHINE, r"SYSTEM\CurrentControlSet\Control\FileSystem" ) as key: data["LongPathsEnabled"] = winreg.QueryValueEx(key, "LongPathsEnabled")[0] except Exception as exc: # noqa: BLE001 data["LongPathsEnabled"] = "unreadable: %s" % exc pwsh = shutil.which("pwsh") if pwsh: policy = run([pwsh, "-NoProfile", "-Command", "Get-ExecutionPolicy -List | Out-String"], root) data["execution_policy"] = policy.stdout.strip() or policy.first_line() else: data["execution_policy"] = "pwsh not found" data["powershell_5_present"] = bool(shutil.which("powershell")) marked = [] tools = root / "tools" if tools.is_dir(): for path in sorted(p for p in tools.iterdir() if p.is_file()): try: with open(str(path) + ":Zone.Identifier", "r", encoding="utf-8", errors="replace") as stream: marked.append({"file": path.name, "zone": stream.read().strip()}) except OSError: pass data["mark_of_the_web"] = marked return data # ------------------------------------------------------------------------ layer 2: stack def scan_tree(root: Path, name: str): """Structure of kb/ or raw/, and the relative paths it holds. No content.""" top = root / name if not top.is_dir(): return {"present": False}, [] files, dirs = [], 0 depth_histogram, longest_rel, longest_abs = {}, 0, 0 counts = { "abs_path_over_240": 0, "abs_path_over_260": 0, "names_with_space": 0, "names_non_ascii": 0, "names_windows_forbidden_chars": 0, "names_trailing_dot_or_space": 0, "names_reserved_device": 0, "names_not_nfc": 0, "case_collisions": 0, } def judge(entry: str) -> None: if " " in entry: counts["names_with_space"] += 1 if any(ord(ch) > 127 for ch in entry): counts["names_non_ascii"] += 1 if WINDOWS_FORBIDDEN.search(entry): counts["names_windows_forbidden_chars"] += 1 if entry.endswith((" ", ".")): counts["names_trailing_dot_or_space"] += 1 if WINDOWS_RESERVED.match(entry): counts["names_reserved_device"] += 1 if unicodedata.normalize("NFC", entry) != entry: counts["names_not_nfc"] += 1 for current, subdirs, names in os.walk(str(top)): entries = sorted(subdirs) + sorted(names) seen = {} for entry in entries: seen.setdefault(entry.casefold(), []).append(entry) judge(entry) counts["case_collisions"] += sum(1 for group in seen.values() if len(group) > 1) dirs += len(subdirs) for entry in names: absolute = os.path.join(current, entry) rel = os.path.relpath(absolute, str(root)).replace(os.sep, "/") files.append(rel) depth = rel.count("/") + 1 depth_histogram[str(depth)] = depth_histogram.get(str(depth), 0) + 1 longest_rel = max(longest_rel, len(rel)) longest_abs = max(longest_abs, len(os.path.abspath(absolute))) if len(os.path.abspath(absolute)) > 240: counts["abs_path_over_240"] += 1 if len(os.path.abspath(absolute)) > 260: counts["abs_path_over_260"] += 1 summary = { "present": True, "files": len(files), "directories": dirs, "depth_histogram": depth_histogram, "longest_relative_path": longest_rel, "longest_absolute_path": longest_abs, } summary.update(counts) return summary, files def collect_stack(root: Path, red: Redactor) -> dict: version = root / "VERSION" configs = {} for path in sorted(root.glob(".wikitool-*.json")): if path.name == ".wikitool-tools.json": continue try: configs[path.name] = red.clean_json(json.loads(read_text(path))) except ValueError: configs[path.name] = {"unparseable": True, "bytes": path.stat().st_size} return { "VERSION": read_text(version).strip() if version.is_file() else "absent", "config_files": configs, "git": _collect_git_state(root), } def _collect_git_state(root: Path) -> dict: quiet = ["git", "-c", "core.quotepath=off"] inside = run(quiet + ["rev-parse", "--is-inside-work-tree"], root) if not inside.ok: return {"repository": False, "error": inside.first_line()} branch = run(quiet + ["rev-parse", "--abbrev-ref", "HEAD"], root) status = run(quiet + ["status", "--porcelain", "-z"], root) log = run(quiet + ["log", "-n", "20", "--name-only", "--format=%x1e%h%x1f%s"], root) return { "repository": True, "branch": branch.stdout.strip() if branch.ok else branch.first_line(), "status": _parse_status(status.stdout) if status.ok else [status.first_line()], "log": _parse_log(log.stdout) if log.ok else [log.first_line()], } def _parse_status(raw: str) -> list: entries = raw.split("\0") lines, i = [], 0 while i < len(entries): entry = entries[i] i += 1 if len(entry) < 4: continue code, path = entry[:2], entry[3:] if code[0] in "RC" and i < len(entries): lines.append("%s %s (from %s)" % (code, path, entries[i])) i += 1 else: lines.append("%s %s" % (code, path)) return lines def _parse_log(raw: str) -> list: lines = [] for block in raw.split("\x1e"): if not block.strip(): continue head, _, rest = block.partition("\n") sha, _, subject = head.partition("\x1f") files = [line for line in rest.splitlines() if line.strip()] if files and all(f.startswith(tuple(d + "/" for d in CONTENT_COMMIT_DIRS)) for f in files): parts = [] for directory in CONTENT_COMMIT_DIRS: count = sum(1 for f in files if f.startswith(directory + "/")) if count: parts.append("%s %d" % (directory, count)) subject = "<content commit: %s files>" % ", ".join(parts) lines.append("%s %s" % (sha.strip(), subject.strip())) return lines # ---------------------------------------------------------------------- layer 3: wikitool def launcher_command(root: Path): tools = root / "tools" if sys.platform == "win32": ps1, sh = tools / "wikitool.ps1", tools / "wikitool" pwsh = shutil.which("pwsh") if ps1.is_file() and pwsh: # Bypass: a policy that blocks the script is a finding for `doctor`'s # execution-policy check, and must not also stop the report from running. return [pwsh, "-NoProfile", "-ExecutionPolicy", "Bypass", "-File", str(ps1)] bash = shutil.which("bash") if sh.is_file() and bash: return [bash, str(sh)] return None launcher = tools / "wikitool" return [str(launcher)] if launcher.is_file() else None def resolve_trace_session(explicit): if explicit: return explicit, "--session" for name in (SESSION_ENV, HARNESS_SESSION_ENV): if os.environ.get(name): return os.environ[name], name return None, None def trace_root(root: Path) -> Path: override = os.environ.get("WIKI_TRACE_DIR") return Path(override) if override else root / "reports" / "telemetry" def collect_trace(root: Path, explicit): """The caller's session trace, or a listing of what exists instead.""" session, source = resolve_trace_session(explicit) base = trace_root(root) if session: path = base / session_slug(session) / "trace.jsonl" if path.is_file(): return {"found": True, "session": session, "source": source, "text": read_text(path)} listing = [] if base.is_dir(): for child in sorted(base.iterdir()): if child.is_dir() and not child.name.startswith("bugreport-"): trace = child / "trace.jsonl" listing.append({"session_directory": child.name, "bytes": trace.stat().st_size if trace.is_file() else 0}) return {"found": False, "session": session, "source": source, "listing": listing} WIKITOOL_CALLS = ( ("version show", ("version", "show"), "inherited"), ("doctor", ("doctor",), "inherited"), ("budget status", ("budget", "status"), "inherited"), ("instructions verify", ("instructions", "verify"), "own"), ("docs verify", ("docs", "verify"), "own"), ) def run_wikitool(root: Path, launcher, own_session: str) -> dict: """Five fixed read commands. `version show` is the startup probe: if it fails, `wikitool` counts as not started and nothing else is run.""" outputs = [] started = False reason = None for label, args, session in WIKITOOL_CALLS: env = dict(os.environ) if session == "own": env[SESSION_ENV] = own_session result = run(launcher + list(args), root, env=env, timeout=TIMEOUT_SECONDS) prefix = "%s=%s " % (SESSION_ENV, own_session) if session == "own" else "" outputs.append({ "label": label, "file": "wikitool/%s.txt" % label.replace(" ", "-"), "session": ("%s (set by the collector)" % own_session) if session == "own" else "inherited from the caller", "text": ( "command: %s%s\nsession: %s\nexit: %s\n%s\n--- stdout ---\n%s\n--- stderr ---\n%s\n" % (prefix, " ".join(launcher + list(args)), "own" if session == "own" else "inherited", result.returncode if result.returncode is not None else result.note, ("note: " + result.note) if result.note and result.returncode is not None else "", result.stdout, result.stderr) ), }) if label == "version show": if not result.ok: reason = "version show %s: %s" % ( ("exit %s" % result.returncode) if result.returncode is not None else result.note, result.first_line()) break started = True return {"started": started, "reason": reason, "outputs": outputs} # ------------------------------------------------------------------------------ bundle class Bundle: def __init__(self, directory: Path) -> None: self.directory = directory self.files = [] def add(self, rel: str, text: str, layer: str, description: str, note: str = "") -> None: write_text(self.directory / rel, text) self.files.append((rel, layer, description, note)) def stamp_now() -> str: return datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ") def unique_stamp(out: Path) -> str: stamp = stamp_now() candidate, n = stamp, 1 while (out / ("bugreport-" + candidate)).exists() or (out / ("bugreport-%s.zip" % candidate)).exists(): n += 1 candidate = "%s-%d" % (stamp, n) return candidate def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str: lines = ["# Bug report bundle", "", "Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"), "Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", STAGE1_NOTICE if args.pseudonymise else PRIVACY_NOTICE, "", "## Options", "", "- chronology: %s" % ("supplied" if args.chronology else "not supplied"), "- transcripts: %d" % len(args.transcript), "- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"), ] + (["- pseudonymisation (`--pseudonymise`): stage 1 applied, stage 2 not yet applied"] if args.pseudonymise else []) + [ "- session trace: %s" % ( "excluded (`--no-trace`)" if args.no_trace else ("included" if trace and trace.get("found") else "not found")), "", "## wikitool", ""] if wikitool is None: lines.append("`wikitool` did not start: no launcher found under `tools/`.") elif wikitool["started"]: lines.append("`wikitool` started. Five read commands ran; outputs are verbatim.") else: lines.append("`wikitool` did not start (%s). The four other commands were not run." % wikitool["reason"]) if wikitool: lines += ["", "| Command | Session id |", "|---|---|"] for out in wikitool["outputs"]: lines.append("| `%s` | %s |" % (out["label"], out["session"])) caller, source = resolve_trace_session(args.session) lines += ["", "Caller session id: %s" % ( "%s (from %s)" % (caller, source) if caller else "not set; `wikitool` falls back to its parent process id"), ""] if trace and not trace.get("found") and not args.no_trace: lines += ["No trace found for the caller's session%s. Existing session directories " "(name and size only):" % ( " `%s`" % trace["session"] if trace.get("session") else ""), ""] for item in trace.get("listing", [])[:100]: lines.append("- `%s` (%d bytes)" % (item["session_directory"], item["bytes"])) if not trace.get("listing"): lines.append("- none") lines.append("") lines += ["## Files", "", "| File | Layer | Contents | Note |", "|---|---|---|---|"] for rel, layer, description, note in sorted(bundle.files, key=lambda f: f[0]): lines.append("| `%s` | %s | %s | %s |" % (rel, layer, description, note)) lines.append("| `MANIFEST.md` | - | this file | |") if gaps: lines += ["", "## Gaps", ""] + ["- %s" % gap for gap in gaps] return "\n".join(lines) + "\n" def final_pass(bundle_dir: Path, red: Redactor, shield, skip) -> None: for path in sorted(bundle_dir.rglob("*")): if not path.is_file(): continue rel = path.relative_to(bundle_dir).as_posix() text = read_text(path) cleaned = red.scrub(text) if shield is not None and rel not in skip: cleaned = shield.apply(cleaned) if cleaned != text: write_text(path, cleaned) def make_zip(bundle_dir: Path, archive: Path) -> None: with zipfile.ZipFile(str(archive), "w", zipfile.ZIP_DEFLATED) as zf: for path in sorted(bundle_dir.rglob("*")): if path.is_file(): zf.write(str(path), bundle_dir.name + "/" + path.relative_to(bundle_dir).as_posix()) def parse_args(argv): parser = argparse.ArgumentParser( prog="bugreport.py", description="Collect a bug-report bundle. Removes secrets, keeps page titles out " "unless --titles is given, uploads nothing.", ) parser.add_argument("--chronology", help="the agent's fact-only chronology (Markdown)") parser.add_argument("--transcript", action="append", default=[], help="a harness transcript to include (repeatable; only on request)") group = parser.add_mutually_exclusive_group() group.add_argument("--session", help="session id whose trace to include") group.add_argument("--no-trace", action="store_true", help="leave the session trace out") parser.add_argument("--titles", action="store_true", help="keep page titles and paths (writes tree-paths.txt)") parser.add_argument("--pseudonymise", action="store_true", help="replace known identities by consistent, shape-preserving " "placeholders (stage 1); the mapping stays beside the bundle") parser.add_argument("--root", help="checkout root (default: parent of tools/)") parser.add_argument("--out", help="output directory (default: <root>/reports)") parser.add_argument("--bundle", help="stage 2: an existing pseudonymised bundle directory") parser.add_argument("--candidates", help="stage 2: file with one further name per line " "('#' starts a comment)") args = parser.parse_args(argv) if bool(args.bundle) != bool(args.candidates): parser.error("--bundle and --candidates belong together") if args.bundle and (args.chronology or args.transcript or args.session or args.no_trace or args.titles or args.pseudonymise or args.root or args.out): parser.error("--bundle/--candidates excludes every collection option") return args def main(argv=None) -> int: args = parse_args(argv) if args.bundle: return run_stage2(Path(args.bundle), Path(args.candidates)) root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent out = Path(args.out).resolve() if args.out else root / "reports" inputs = [("chronology", args.chronology)] + [("transcript", t) for t in args.transcript] for label, value in inputs: if value and not Path(value).is_file(): sys.stderr.write("bugreport: %s file not found: %s\n" % (label, value)) return 1 try: out.mkdir(parents=True, exist_ok=True) stamp = unique_stamp(out) bundle_dir = out / ("bugreport-" + stamp) bundle_dir.mkdir() return _collect(args, root, out, stamp, bundle_dir) except OSError as exc: sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc) return 1 def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int: red = Redactor() bundle = Bundle(bundle_dir) gaps = [] trace = None if not args.no_trace: trace = guarded(collect_trace, root, args.session) if trace.get("found"): bundle.add("trace.jsonl", trace["text"], "3", "session trace of the caller", MAY_CONTAIN_CONTENT) bundle.add("environment.json", dump_json(guarded(collect_environment, root, red)), "1", "OS, Python, shells, harness, PATH, environment variables (names; values for a " "fixed list), git configuration, venv, line endings, Windows details") stack = guarded(collect_stack, root, red) bundle.add("stack.json", dump_json(stack), "2", "VERSION, `.wikitool-*.json` (secrets removed), git status and log") content_paths = [] tree = {} for name in CONTENT_DIRS: summary, files = _safe_scan(root, name) tree[name] = summary content_paths.extend(files) bundle.add("tree-structure.json", dump_json(tree), "2", "kb/ and raw/: counts, depth, path lengths, name problems - no content") if args.titles: bundle.add("tree-paths.txt", "\n".join(sorted(content_paths)) + "\n", "2", "all relative paths under kb/ and raw/", "contains page titles") wikitool = None launcher = launcher_command(root) if launcher: wikitool = run_wikitool(root, launcher, "bugreport-" + stamp) for item in wikitool["outputs"]: bundle.add(item["file"], item["text"], "3", "verbatim output of `%s`" % item["label"]) else: gaps.append("no launcher found under tools/, so no wikitool command ran") if args.chronology: bundle.add("CHRONOLOGY.md", read_text(Path(args.chronology)), "4", "the agent's chronology", MAY_CONTAIN_CONTENT) else: gaps.append("no chronology was supplied") used = set() for source in args.transcript: name = Path(source).name target, n = name, 1 while target in used: n += 1 target = "%d-%s" % (n, name) used.add(target) bundle.add("transcripts/" + target, read_text(Path(source)), "4", "harness transcript", MAY_CONTAIN_CONTENT) write_text(bundle_dir / "MANIFEST.md", build_manifest(args, root, stamp, bundle, wikitool, trace, gaps)) shield = None if not args.titles: shield = TitleShield( rel for rel in content_paths if rel.rsplit("/", 1)[-1] not in STACK_NAMES ) final_pass(bundle_dir, red, shield, {"tree-paths.txt"}) pz = None if args.pseudonymise: pz = Pseudonymiser() for original, kind in collect_identities(root): pz.add_identity(original, kind) pz.finish() pseudonymise_bundle(bundle_dir, pz) manifest_path = bundle_dir / "MANIFEST.md" write_text(manifest_path, replace_section( read_text(manifest_path), "Pseudonyms", pseudonym_table(pz))) write_text(_sibling(bundle_dir, ".pseudonyms.json"), dump_json(pz.to_json(bundle_dir.name))) write_text(_sibling(bundle_dir, ".review.txt"), build_review(bundle_dir, pz)) archive = out / ("bugreport-%s.zip" % stamp) make_zip(bundle_dir, archive) print("Bundle: %s" % bundle_dir) print("Archive: %s" % archive) if pz is not None: print("Mapping: %s" % _sibling(bundle_dir, ".pseudonyms.json")) print("Review: %s" % _sibling(bundle_dir, ".review.txt")) print("Both hold originals and stay on this machine: never share them.") print("Stage 2: python3 tools/bugreport.py --bundle %s --candidates <file>" % bundle_dir) print() print(STAGE1_NOTICE if pz is not None else PRIVACY_NOTICE) return 0 def _sibling(bundle_dir: Path, suffix: str) -> Path: return bundle_dir.with_name(bundle_dir.name + suffix) def parse_candidates(text: str): names = [] for line in text.splitlines(): line = re.sub(r"(^|\s)#.*$", "", line).strip() if line and line not in names: names.append(line) return names def run_stage2(bundle_dir: Path, candidates: Path) -> int: bundle_dir = bundle_dir.resolve() mapping = _sibling(bundle_dir, ".pseudonyms.json") problem = None if not bundle_dir.is_dir(): problem = "bundle directory not found: %s" % bundle_dir elif not candidates.is_file(): problem = "candidates file not found: %s" % candidates elif not mapping.is_file(): problem = ("no mapping beside the bundle (%s): stage 2 needs the salt stage 1 used, so " "collect the report again with --pseudonymise" % mapping.name) if problem: sys.stderr.write("bugreport: %s\n" % problem) return 1 try: pz = Pseudonymiser.from_json(json.loads(read_text(mapping))) except (ValueError, KeyError) as exc: sys.stderr.write("bugreport: the mapping cannot be read: %s\n" % exc) return 1 files = text_files(bundle_dir) corpus = "\n".join(read_text(path) for path in files) applied, skipped = [], [] placeholders = pz.placeholder_words() for name in parse_candidates(read_text(candidates)): words = _words(name) if len(name) < MIN_IDENTITY or not words: skipped.append((name, "under %d characters" % MIN_IDENTITY)) elif all(w in VOCABULARY or w in pz.keep for w in words): skipped.append((name, "system vocabulary")) elif all(w in placeholders for w in words): skipped.append((name, "already a placeholder")) elif pz.find(corpus, name) == 0: skipped.append((name, "not found in the bundle")) else: applied.append(name) try: for name in applied: pz.add_identity(name, "stage-2 candidate", stage=2) pz.stage = 2 pseudonymise_bundle(bundle_dir, pz, stage=2) manifest_path = bundle_dir / "MANIFEST.md" manifest = read_text(manifest_path) manifest = replace_section(manifest, "Privacy", RESIDUAL_NOTICE) manifest = replace_section(manifest, "Pseudonyms", pseudonym_table(pz)) manifest = re.sub(r"(?m)^- pseudonymisation \(`--pseudonymise`\): .*$", "- pseudonymisation (`--pseudonymise`): stage 1 and stage 2 applied", manifest) write_text(manifest_path, manifest) write_text(mapping, dump_json(pz.to_json(bundle_dir.name))) archive = bundle_dir.with_name(bundle_dir.name + ".zip") temporary = archive.with_name(archive.name + ".tmp") make_zip(bundle_dir, temporary) os.replace(str(temporary), str(archive)) except OSError as exc: sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc) return 1 print("Bundle: %s" % bundle_dir) print("Archive: %s" % archive) print("Mapping: %s (holds originals; delete it after the last stage 2 run)" % mapping) for name in applied: print("applied: %s" % name) for name, reason in skipped: print("not applied: %s - %s" % (name, reason)) print() print(RESIDUAL_NOTICE) return 0 def _safe_scan(root: Path, name: str): try: return scan_tree(root, name) except Exception as exc: # noqa: BLE001 return {"collector_error": "%s: %s" % (type(exc).__name__, exc)}, [] if __name__ == "__main__": sys.exit(main())