Files
torben 5d937233b1
CI / verify (push) Successful in 5m20s
CI / pwsh (push) Successful in 1m53s
Release / release (push) Successful in 36s
fix: tools/bugreport launcher finds a Python 3.8+ itself and never starts a Store alias (#166)
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2

Files changed:
- .gitea/workflows/ci.yml
- CHANGES.md
- INSTALL.md
- VERSION
- instructions/bug-report.md
- tools/README.md
- tools/bugreport
- tools/bugreport.ps1
- tools/bugreport.py
- tools/chemenu/tests/test_bugreport_launcher.py
- tools/chemenu/tests/test_instructions_shell.py
- tools/chemenu/tests/test_portability.py
2026-10-02 13:12:19 +02:00

1436 lines
56 KiB
Python

#!/usr/bin/env python3
"""Collect a bug-report bundle from this checkout.
Runs on the base Python with the standard library only, and imports nothing from
`chemenu`: the case it exists for is a checkout where `wikitool` does not start
- no venv, a broken package, a Python that is too old. Syntax stays at Python
3.8 so that even an old interpreter can still produce a report.
tools/bugreport [--chronology FILE] [--transcript FILE]...
[--session ID | --no-trace] [--titles]
[--pseudonymise] [--root DIR] [--out DIR]
tools/bugreport --bundle DIR --candidates FILE
`tools/bugreport` (and `tools/bugreport.ps1` under PowerShell) finds a Python 3.8
or newer and runs this file with it, skipping the Microsoft Store's aliases that
`python3` resolves to on Windows. Started directly - `<python> tools/bugreport.py`
- it works the same, with whichever interpreter was named.
The bundle is `<out>/bugreport-<UTC stamp>/` plus a zip beside it, in four
layers: the environment, the stack, what `wikitool` prints (if it starts), and
the agent's chronology. Secrets are always removed. Page titles are kept out of
everything this script generates unless `--titles` is given. With `--pseudonymise`
known identities are replaced by consistent, shape-preserving placeholders (stage 1);
`--bundle`/`--candidates` applies the names a model found beyond them (stage 2). The
mapping stays beside the bundle, never in it. Nothing is uploaded.
Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script
opens no gate. See instructions/bug-report.md.
"""
from __future__ import annotations
import argparse
import datetime
import getpass
import hashlib
import hmac
import json
import os
import platform
import re
import secrets
import shutil
import socket
import subprocess
import sys
import unicodedata
import zipfile
from pathlib import Path
TIMEOUT_SECONDS = 300
PROBE_TIMEOUT_SECONDS = 20
REMOVED = "<removed>"
NOT_COLLECTED = "<not collected>"
SESSION_ENV = "WIKITOOL_SESSION_ID"
HARNESS_SESSION_ENV = "CLAUDE_CODE_SESSION_ID"
SECRET_NAME = re.compile(r"token|password|secret|key|auth", re.IGNORECASE)
HARNESS_MARKERS = (
(HARNESS_SESSION_ENV, "claude-code"),
)
ENV_ALLOW_EXACT = frozenset(
{
"PATH", "PATHEXT", "SHELL", "TERM", "TERM_PROGRAM", "TERM_PROGRAM_VERSION",
"COLORTERM", "LANG", "LANGUAGE", "VIRTUAL_ENV", "MSYSTEM", "COMSPEC",
"PSMODULEPATH",
}
| {name for name, _ in HARNESS_MARKERS}
)
ENV_ALLOW_PREFIX = ("LC_", "PYTHON", "WIKI_", "WIKITOOL_", "CHEMENU_")
# Names under kb/ that the stack generates or owns: not titles, and unreadable
# in a bundle if they are masked.
STACK_NAMES = frozenset(
{
"CONTRACT.md", "CONVENTIONS.md", "CONVENTIONS.md.template", "COLLECTION.md",
"COLLECTION.md.template", "INDEX.md", "index.md", "log.md", "provenance.md",
".gitkeep",
}
)
CONTENT_DIRS = ("kb", "raw")
CONTENT_COMMIT_DIRS = ("kb", "raw", "work")
MAY_CONTAIN_CONTENT = "may contain page content and titles"
PRIVACY_NOTICE = (
"This bundle is not pseudonymised. It contains private data: machine, user and path "
"names, PATH entries, git remotes and commit subjects and - if included - the session "
"trace, the chronology and transcripts, which may contain page content and titles. "
"Secrets are removed, but read the bundle before you share it. Choose the channel "
"yourself; neither this tool nor the agent uploads anything."
)
STAGE1_NOTICE = (
"Pseudonymisation: stage 1 applied, stage 2 not yet applied. Identities this script could read "
"from the machine (user, host, home and repository paths, git identity, remotes) are replaced by "
"consistent placeholders that keep length, spaces, hyphens, character classes, separators and depth. "
"Names it cannot know - people, companies, customers, internal hosts, projects - may remain, above "
"all in the trace, the chronology and transcripts. Secrets are removed. Read the bundle before you "
"share it. Choose the channel yourself; neither this tool nor the agent uploads anything."
)
RESIDUAL_NOTICE = (
"Pseudonymisation: stage 1 and stage 2 applied. Stage 2 is the judgement of a model: it can miss "
"names, companies, hosts and projects, above all in the free text of a trace or transcript over "
"100 KB, which the model did not read in full. The forms - lengths, character classes, separators "
"and depth - are kept on purpose, so a rare name can still be recognisable by its shape. Read the "
"bundle before you share it. Choose the channel yourself; neither this tool nor the agent uploads "
"anything."
)
PUBLIC_ORIGIN = "gitea.nehmer.net/torben/chemenu" # same origin as chemenu/version.py; this script imports nothing from chemenu
MIN_IDENTITY = 3
NON_ASCII_LOWER = "äöüéèêàáâçñõøåæ"
WORD = re.compile(r"[^\W_]+")
VOCABULARY = frozenset(
w.lower() for w in (
"Windows System32 SysWOW64 Program Files ProgramData Users Public AppData Local LocalLow "
"Roaming Temp Microsoft WindowsApps Programs Documents Desktop Downloads OneDrive "
"DESKTOP LAPTOP "
"home usr local bin opt etc var tmp mnt src Library Applications "
"Python Git Scripts venv chocolatey "
"chemenu wikitool tools reports kb raw bugreport "
"GmbH AG KG SE Inc Ltd LLC "
"removed collected title depth len space "
"json jsonl md txt py ps1 zip cfg template"
).split()
)
WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE)
WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]')
WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]")
URL_USERINFO = re.compile(r"(?P<scheme>[A-Za-z][A-Za-z0-9+.\-]*)://(?P<userinfo>[^/\s@\"'<>]+)@")
AUTH_HEADER = re.compile(r"(?im)(authorization\s*[:=]\s*)(?:(?:bearer|basic|token)\s+)?\S+")
BEARER = re.compile(r"(?i)\b(bearer)\s+[A-Za-z0-9._~+/=\-]{8,}")
# --------------------------------------------------------------------------- secrets
class Redactor:
"""Every secret value found while collecting, replaced again in a final pass."""
def __init__(self) -> None:
self.values = set()
def add(self, value) -> None:
if isinstance(value, str) and len(value) >= 4:
self.values.add(value)
def add_any(self, node) -> None:
if isinstance(node, dict):
for item in node.values():
self.add_any(item)
elif isinstance(node, (list, tuple)):
for item in node:
self.add_any(item)
else:
self.add(node)
def scrub_urls(self, text: str) -> str:
def repl(match):
scheme, userinfo = match.group("scheme"), match.group("userinfo")
if scheme.lower() in ("http", "https"):
self.add(userinfo)
for part in userinfo.split(":"):
self.add(part)
return scheme + "://"
if ":" in userinfo:
user, password = userinfo.split(":", 1)
self.add(password)
return scheme + "://" + user + "@"
return match.group(0)
return URL_USERINFO.sub(repl, text)
def scrub(self, text: str) -> str:
text = self.scrub_urls(text)
text = AUTH_HEADER.sub(lambda m: m.group(1) + REMOVED, text)
text = BEARER.sub(lambda m: m.group(1) + " " + REMOVED, text)
forms = set()
for value in self.values:
forms.add(value)
escaped = json.dumps(value, ensure_ascii=False)[1:-1]
if escaped != value:
forms.add(escaped)
for form in sorted(forms, key=len, reverse=True):
text = text.replace(form, REMOVED)
return text
def clean_json(self, node):
if isinstance(node, dict):
out = {}
for key, value in node.items():
if SECRET_NAME.search(str(key)):
self.add_any(value)
out[key] = REMOVED
else:
out[key] = self.clean_json(value)
return out
if isinstance(node, list):
return [self.clean_json(item) for item in node]
if isinstance(node, str):
return self.scrub_urls(node)
return node
# ------------------------------------------------------------------------ title shield
def _shape(rel: str) -> str:
parts = rel.split("/")
name = parts[-1]
stem, dot, suffix = name.rpartition(".")
flags = ["depth=%d" % len(parts), "len=%d" % len(rel)]
if " " in rel:
flags.append("space")
if any(ord(ch) > 127 for ch in rel):
flags.append("non-ascii")
return "%s/<%s>%s" % (parts[0], " ".join(flags), (dot + suffix) if dot and stem else "")
class TitleShield:
"""Replaces the relative path of a file under kb/ or raw/ with its shape.
A title that stands as bare prose is not found, on purpose: replacing titles
word by word would hit a title like "Git" everywhere in the bundle.
"""
def __init__(self, rel_paths) -> None:
self.forms = {}
for rel in rel_paths:
shape = _shape(rel)
for form in (rel, rel.replace("/", "\\"), rel.replace("/", "\\\\")):
self.forms[form] = shape
self.lengths = sorted({len(form) for form in self.forms}, reverse=True)
self.start = re.compile(r"(?:%s)(?:/|\\)" % "|".join(CONTENT_DIRS))
def apply(self, text: str) -> str:
if self.forms:
out, last = [], 0
for match in self.start.finditer(text):
i = match.start()
if i < last:
continue
for length in self.lengths:
shape = self.forms.get(text[i:i + length])
if shape:
out.append(text[last:i])
out.append(shape)
last = i + length
break
out.append(text[last:])
text = "".join(out)
return WIKILINK.sub(lambda m: "[[<title %s>]]" % _title_flags(m.group(1)), text)
def _title_flags(title: str) -> str:
flags = ["len=%d" % len(title)]
if " " in title:
flags.append("space")
if any(ord(ch) > 127 for ch in title):
flags.append("non-ascii")
return " ".join(flags)
# ------------------------------------------------------------------------ pseudonymisation
def _words(text: str):
return [m.group(0).lower() for m in WORD.finditer(text)]
class Pseudonymiser:
"""Consistent, shape-preserving replacement of identities, word by word.
A placeholder hangs on the word, not on the identity: "Beispiel" is replaced the
same way inside "OneDrive - Beispiel GmbH" (stage 1) and inside the candidate
"Beispiel GmbH" (stage 2), so the two stages agree without coordinating.
"""
def __init__(self, salt=None) -> None:
self.salt = salt or secrets.token_hex(16)
self.words = {}
self.keep = set()
self.entries = []
self.skipped = 0
self.stage = 1
self._index = {}
self._taken = set(VOCABULARY)
self._regex = {}
self._forms = {}
# -- persistence
def to_json(self, bundle_name: str) -> dict:
return {
"schema": 1, "bundle": bundle_name, "stage": self.stage, "salt": self.salt,
"keep": sorted(self.keep), "skipped": self.skipped, "words": self.words,
"entries": self.entries,
}
@classmethod
def from_json(cls, data: dict) -> "Pseudonymiser":
pz = cls(data["salt"])
pz.stage = data.get("stage", 1)
pz.keep = set(data.get("keep", []))
pz.skipped = data.get("skipped", 0)
pz.words = dict(data.get("words", {}))
pz.entries = list(data.get("entries", []))
for key, pseudo in pz.words.items():
pz._taken.add(key)
pz._taken.add(pseudo)
for index, entry in enumerate(pz.entries):
pz._index[entry["original"].lower()] = index
pz._taken.update(_words(entry["original"]))
return pz
# -- placeholders
def _stream(self, key: str, counter: int):
block = 0
salt = bytes.fromhex(self.salt)
while True:
message = "%s\x00%d\x00%d" % (key, counter, block)
for byte in hmac.new(salt, message.encode("utf-8"), hashlib.sha256).digest():
yield byte
block += 1
def pseudo_word(self, word: str) -> str:
key = word.lower()
if key in VOCABULARY or key in self.keep:
return word
base = self.words.get(key)
if base is None:
counter = 0
while True:
stream = self._stream(key, counter)
chars = []
for char in word:
byte = next(stream)
if char.isdigit():
chars.append(str(byte % 10))
elif char.isascii():
chars.append(chr(ord("a") + byte % 26))
else:
chars.append(NON_ASCII_LOWER[byte % len(NON_ASCII_LOWER)])
base = "".join(chars)
if (base != key and base not in self._taken) or counter >= 500:
break
counter += 1
self.words[key] = base
self._taken.add(base)
if len(base) != len(word):
base = (base * (len(word) // max(len(base), 1) + 1))[:len(word)]
return "".join(
b.upper() if orig.isupper() and not b.isdigit() else b for orig, b in zip(word, base)
)
def pseudo_text(self, text: str) -> str:
return WORD.sub(lambda m: self.pseudo_word(m.group(0)), text)
# -- identities
def add_identity(self, original: str, kind: str, stage: int = 1) -> bool:
original = original.strip()
if not original or original.lower() in self._index:
return False
words = _words(original)
if len(original) < MIN_IDENTITY or not words:
self.skipped += 1
return False
if kind in ("host", "domain", "remote") and "." in original:
label = _words(original)[-1]
if not label.isdigit():
self.keep.add(label)
if all(w in VOCABULARY or w in self.keep for w in words):
self.skipped += 1
return False
self._taken.update(words)
self._index[original.lower()] = len(self.entries)
self.entries.append({"original": original, "placeholder": None, "kind": kind,
"stage": stage, "count": 0})
self._regex = {}
return True
def finish(self) -> None:
for entry in self.entries:
if entry["placeholder"] is None:
entry["placeholder"] = self.pseudo_text(entry["original"])
# -- applying
def _build(self, stage) -> None:
self.finish()
forms = {PUBLIC_ORIGIN.lower(): (None, False)}
for entry in self.entries:
if stage is not None and entry["stage"] != stage:
continue
raw = entry["original"]
forms.setdefault(raw.lower(), (entry, False))
escaped = json.dumps(raw, ensure_ascii=True)[1:-1]
if escaped.lower() != raw.lower():
forms.setdefault(escaped.lower(), (entry, True))
names = sorted(forms, key=len, reverse=True)
self._regex[stage] = (forms, re.compile(
r"(?<![^\W_])(?:%s)(?![^\W_])" % "|".join(re.escape(n) for n in names), re.IGNORECASE
))
def apply(self, text: str, stage=None) -> str:
if not any(stage is None or e["stage"] == stage for e in self.entries):
return text
if stage not in self._regex:
self._build(stage)
forms, regex = self._regex[stage]
def repl(match):
found = forms.get(match.group(0).lower())
if found is None or found[0] is None:
return match.group(0)
entry, escaped = found
if escaped:
try:
decoded = json.loads('"%s"' % match.group(0))
except ValueError:
return match.group(0)
entry["count"] += 1
return json.dumps(self.pseudo_text(decoded), ensure_ascii=True)[1:-1]
entry["count"] += 1
return self.pseudo_text(match.group(0))
return regex.sub(repl, text)
def find(self, text: str, original: str) -> int:
raw = re.escape(original)
escaped = re.escape(json.dumps(original, ensure_ascii=True)[1:-1])
pattern = re.compile(r"(?<![^\W_])(?:%s|%s)(?![^\W_])" % (raw, escaped), re.IGNORECASE)
return len(pattern.findall(text))
def placeholder_words(self) -> set:
return set(self.words.values())
def _path_parts(value: str):
return [part for part in re.split(r"[\\/]+", value) if part]
def _remote_identities(url: str):
"""(original, kind) pairs for one remote URL. The stack's public origin is exempt."""
url = url.strip()
if not url or PUBLIC_ORIGIN in url.lower().replace("\\", "/"):
return []
found = []
scheme = re.match(r"^[A-Za-z][A-Za-z0-9+.\-]*://", url)
scp = re.match(r"^(?:[^@/\\:\s]+@)?([^/\\:\s]{2,}):(?![/\\])(.*)$", url)
if scheme:
rest = url[scheme.end():]
netloc, _, path = rest.partition("/")
host = netloc.rsplit("@", 1)[-1].split(":")[0]
found.append((host, "remote"))
parts = _path_parts(path)
elif scp:
found.append((scp.group(1), "remote"))
parts = _path_parts(scp.group(2))
else:
parts = _path_parts(url)
found.extend((part, "remote") for part in parts)
return found
def collect_identities(root: Path):
"""Identities readable from this machine, in the order they are registered."""
found = []
user = set()
try:
user.add(getpass.getuser())
except Exception: # noqa: BLE001 - no account name is not a failure
pass
for name in ("USER", "USERNAME", "LOGNAME"):
if os.environ.get(name):
user.add(os.environ[name])
found.extend((value, "user name") for value in sorted(user))
if os.environ.get("USERDOMAIN"):
found.append((os.environ["USERDOMAIN"], "domain"))
if os.environ.get("COMPUTERNAME"):
found.append((os.environ["COMPUTERNAME"], "host"))
try:
found.append((socket.gethostname(), "host"))
except OSError:
pass
found.extend((part, "home path") for part in _path_parts(os.path.expanduser("~")))
found.extend((part, "repo path") for part in _path_parts(str(root)))
for key in ("user.name", "user.email"):
result = run(["git", "config", "--get", key], root)
value = result.stdout.strip() if result.ok else ""
if value:
found.append((value, "git identity"))
if key == "user.email" and "@" in value:
local, _, domain = value.rpartition("@")
found.append((local, "git identity"))
found.append((domain, "domain"))
remotes = run(["git", "remote", "-v"], root)
if remotes.ok:
for line in remotes.stdout.splitlines():
fields = line.split("\t", 1)
if len(fields) == 2:
found.extend(_remote_identities(re.sub(r"\s+\((?:fetch|push)\)\s*$", "", fields[1])))
return found
def text_files(bundle_dir: Path):
return [p for p in sorted(bundle_dir.rglob("*")) if p.is_file()]
def pseudonymise_bundle(bundle_dir: Path, pz: Pseudonymiser, stage=None) -> None:
for path in text_files(bundle_dir):
text = read_text(path)
changed = pz.apply(text, stage)
if changed != text:
write_text(path, changed)
EMAIL_RE = re.compile(r"[\w.+\-]+@[\w\-]+(?:\.[\w\-]+)+")
URL_RE = re.compile(r"[A-Za-z][A-Za-z0-9+.\-]*://[^\s\"'<>)\]]+")
PATH_RE = re.compile(r"(?:[A-Za-z]:)?(?:[\\/]+[^\\/\s\"'<>|:*?,;()\[\]{}]+){2,}")
def build_review(bundle_dir: Path, pz: Pseudonymiser) -> str:
"""What stage 1 leaves structured and unmasked, for the model that does stage 2."""
found = {}
def note(kind: str, value: str, rel: str) -> None:
words = _words(value)
if len(value) < MIN_IDENTITY or not words:
return
if all(w in VOCABULARY or w in pz.keep or w in pz.placeholder_words() for w in words):
return
entry = found.setdefault((kind, value), {"count": 0, "files": []})
entry["count"] += 1
if rel not in entry["files"]:
entry["files"].append(rel)
for path in text_files(bundle_dir):
rel = path.relative_to(bundle_dir).as_posix()
text = read_text(path)
for match in EMAIL_RE.finditer(text):
note("mail address", match.group(0), rel)
for match in URL_RE.finditer(text):
rest = match.group(0).split("://", 1)[1]
netloc, _, tail = rest.partition("/")
note("host", netloc.rsplit("@", 1)[-1], rel)
for part in _path_parts(tail):
note("url segment", part, rel)
for match in PATH_RE.finditer(text):
for part in _path_parts(match.group(0)):
note("path component", part, rel)
lines = [
"# Review list for stage 2 - local only, never part of the bundle. It holds what stage 1",
"# left in the bundle, which can still be a real name. One line: kind, count, value, files.",
]
for (kind, value), info in sorted(found.items(), key=lambda kv: (-kv[1]["count"], kv[0])):
lines.append("%s\t%d\t%s\t%s" % (kind, info["count"], value, ", ".join(info["files"][:5])))
return "\n".join(lines) + "\n"
def pseudonym_table(pz: Pseudonymiser) -> str:
lines = ["| Placeholder | Kind | Stage | Occurrences |", "|---|---|---|---|"]
for entry in pz.entries:
placeholder = entry["placeholder"].replace("|", "\\|")
lines.append("| `%s` | %s | %d | %d |" % (
placeholder, entry["kind"], entry["stage"], entry["count"]))
lines += ["", "%d identities were left unchanged (under %d characters, or system vocabulary); "
"their values are not listed." % (pz.skipped, MIN_IDENTITY)]
return "\n".join(lines)
def replace_section(manifest: str, heading: str, body: str) -> str:
sections = re.split(r"(?m)^(?=## )", manifest)
block = "## %s\n\n%s\n" % (heading, body)
for index, section in enumerate(sections):
if section.startswith("## %s\n" % heading):
sections[index] = block
break
else:
sections.append(block)
return "\n\n".join(part.strip("\n") for part in sections if part.strip()) + "\n"
# ---------------------------------------------------------------------------- helpers
class Result:
def __init__(self, returncode, stdout, stderr, note=None) -> None:
self.returncode = returncode
self.stdout = stdout
self.stderr = stderr
self.note = note
@property
def ok(self) -> bool:
return self.returncode == 0
def first_line(self) -> str:
for text in (self.stderr, self.stdout, self.note or ""):
for line in text.splitlines():
if line.strip():
return line.strip()
return ""
def _decode(data) -> str:
if data is None:
return ""
if isinstance(data, str):
return data
return data.decode("utf-8", errors="replace")
def run(cmd, cwd, env=None, timeout=PROBE_TIMEOUT_SECONDS) -> Result:
try:
proc = subprocess.run(
cmd, cwd=str(cwd), env=env, stdin=subprocess.DEVNULL,
stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout,
)
except subprocess.TimeoutExpired as exc:
return Result(None, _decode(exc.stdout), _decode(exc.stderr),
"timed out after %d s" % timeout)
except OSError as exc:
return Result(None, "", "", "could not start: %s" % exc)
return Result(proc.returncode, _decode(proc.stdout), _decode(proc.stderr))
def read_text(path: Path) -> str:
with open(str(path), "r", encoding="utf-8", errors="replace", newline="") as handle:
return handle.read()
def write_text(path: Path, text: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
with open(str(path), "w", encoding="utf-8", newline="\n") as handle:
handle.write(text)
def dump_json(data) -> str:
return json.dumps(data, indent=2, ensure_ascii=False) + "\n"
def guarded(fn, *args):
"""A broken layer must not take the whole report down."""
try:
return fn(*args)
except Exception as exc: # noqa: BLE001 - the report is the point
return {"collector_error": "%s: %s" % (type(exc).__name__, exc)}
def session_slug(value: str) -> str:
return re.sub(r"[^A-Za-z0-9._-]+", "__", value).strip("_") or "unknown"
# ------------------------------------------------------------------- layer 1: environment
def _version_line(cmd, cwd) -> str:
result = run(cmd, cwd)
return result.first_line() if result.ok or result.first_line() else (result.note or "")
def collect_environment(root: Path, red: Redactor) -> dict:
data = {
"os": {
"platform": platform.platform(),
"system": platform.system(),
"release": platform.release(),
"version": platform.version(),
"machine": platform.machine(),
},
"python": _collect_python(root),
"shells": _collect_shells(root),
"harness": _collect_harness(),
"path_entries": [
{"entry": entry, "exists": os.path.isdir(entry)}
for entry in os.environ.get("PATH", "").split(os.pathsep) if entry
],
"env": _collect_env(red),
"tools_config": _collect_tools_config(root, red),
"git": _collect_git_config(root, red),
"venv": _collect_venv(root, red),
"line_endings": _collect_line_endings(root),
"gitattributes": _read_optional(root / ".gitattributes", red),
"windows": _collect_windows(root),
}
return data
def _collect_python(root: Path) -> dict:
running = {
"executable": sys.executable,
"version": sys.version.replace("\n", " "),
"version_info": list(sys.version_info[:3]),
}
candidates = []
for name, extra in (("python3", []), ("python", []), ("py", ["-3"])):
path = shutil.which(name)
entry = {"name": " ".join([name] + extra), "path": path}
if path:
entry["version"] = _version_line([path] + extra + ["--version"], root)
entry["windows_store_alias"] = "windowsapps" in path.lower()
candidates.append(entry)
return {"running": running, "candidates": candidates}
def _collect_shells(root: Path) -> dict:
shells = {"SHELL": os.environ.get("SHELL"), "COMSPEC": os.environ.get("COMSPEC")}
for name in ("bash", "zsh", "fish"):
path = shutil.which(name)
shells[name] = {"path": path,
"version": _version_line([path, "--version"], root) if path else None}
for name in ("pwsh", "powershell"):
path = shutil.which(name)
shells[name] = {
"path": path,
"version": _version_line(
[path, "-NoProfile", "-Command", "$PSVersionTable.PSVersion.ToString()"], root
) if path else None,
}
return shells
def _collect_harness() -> dict:
detected = [
{"marker": name, "harness": harness}
for name, harness in HARNESS_MARKERS if os.environ.get(name)
]
if os.environ.get("TERM_PROGRAM") == "vscode":
detected.append({"marker": "TERM_PROGRAM=vscode", "harness": "vscode terminal"})
return {"detected": detected or "unrecognised"}
def _env_allowed(name: str) -> bool:
upper = name.upper()
return upper in ENV_ALLOW_EXACT or upper.startswith(ENV_ALLOW_PREFIX)
def _collect_env(red: Redactor) -> dict:
env = {}
for name in sorted(os.environ):
value = os.environ[name]
if SECRET_NAME.search(name):
red.add(value)
env[name] = REMOVED if _env_allowed(name) else NOT_COLLECTED
elif _env_allowed(name):
env[name] = value
else:
env[name] = NOT_COLLECTED
return env
def _collect_tools_config(root: Path, red: Redactor) -> dict:
path = root / ".wikitool-tools.json"
if not path.is_file():
return {"status": "absent"}
try:
parsed = json.loads(read_text(path))
except ValueError as exc:
return {"status": "unparseable", "error": str(exc)}
checks = []
def walk(node, where):
if isinstance(node, dict):
for key, value in node.items():
walk(value, where + [str(key)])
elif isinstance(node, list):
for index, value in enumerate(node):
walk(value, where + [str(index)])
elif isinstance(node, str) and os.path.isabs(node):
checks.append({"at": ".".join(where), "path": node, "exists": os.path.exists(node)})
walk(parsed, [])
return {"status": "present", "content": red.clean_json(parsed), "path_checks": checks}
def _collect_git_config(root: Path, red: Redactor) -> dict:
version = run(["git", "--version"], root)
if not version.ok:
return {"available": False, "error": version.note or version.first_line()}
config = {}
for key in ("core.autocrlf", "core.longpaths", "core.filemode"):
result = run(["git", "config", "--get", key], root)
config[key] = result.stdout.strip() if result.ok else "(unset)"
name = run(["git", "config", "--get", "user.name"], root)
config["user.name is set"] = bool(name.ok and name.stdout.strip())
remotes = run(["git", "remote", "-v"], root)
return {
"available": True,
"version": version.stdout.strip(),
"config": config,
"remotes": red.scrub_urls(remotes.stdout).splitlines() if remotes.ok
else ["(not a repository, or git failed: %s)" % remotes.first_line()],
}
def _collect_venv(root: Path, red: Redactor) -> dict:
venv = root / "tools" / ".venv"
if not venv.is_dir():
return {"present": False}
layout = ("bin" if (venv / "bin").is_dir() else None) or (
"Scripts" if (venv / "Scripts").is_dir() else None)
cfg = venv / "pyvenv.cfg"
return {
"present": True,
"layout": layout or "unknown",
"python_found": any(
(venv / sub / exe).exists()
for sub, exe in (("bin", "python"), ("Scripts", "python.exe"))
),
"pyvenv.cfg": read_text(cfg) if cfg.is_file() else None,
}
def _collect_line_endings(root: Path) -> dict:
tools = root / "tools"
files = [tools / "wikitool"] + (sorted(tools.glob("*.ps1")) if tools.is_dir() else [])
result = {}
for path in files:
if not path.is_file():
continue
raw = path.read_bytes()
crlf = raw.count(b"\r\n")
result[path.name] = {"crlf": crlf, "lf_only": raw.count(b"\n") - crlf}
return result
def _read_optional(path: Path, red: Redactor):
return red.scrub_urls(read_text(path)) if path.is_file() else None
def _collect_windows(root: Path) -> dict:
if sys.platform != "win32":
return {"applicable": False}
data = {"applicable": True}
try:
import winreg # noqa: PLC0415 - only exists on Windows
with winreg.OpenKey(
winreg.HKEY_LOCAL_MACHINE, r"SYSTEM\CurrentControlSet\Control\FileSystem"
) as key:
data["LongPathsEnabled"] = winreg.QueryValueEx(key, "LongPathsEnabled")[0]
except Exception as exc: # noqa: BLE001
data["LongPathsEnabled"] = "unreadable: %s" % exc
pwsh = shutil.which("pwsh")
if pwsh:
policy = run([pwsh, "-NoProfile", "-Command", "Get-ExecutionPolicy -List | Out-String"], root)
data["execution_policy"] = policy.stdout.strip() or policy.first_line()
else:
data["execution_policy"] = "pwsh not found"
data["powershell_5_present"] = bool(shutil.which("powershell"))
marked = []
tools = root / "tools"
if tools.is_dir():
for path in sorted(p for p in tools.iterdir() if p.is_file()):
try:
with open(str(path) + ":Zone.Identifier", "r", encoding="utf-8",
errors="replace") as stream:
marked.append({"file": path.name, "zone": stream.read().strip()})
except OSError:
pass
data["mark_of_the_web"] = marked
return data
# ------------------------------------------------------------------------ layer 2: stack
def scan_tree(root: Path, name: str):
"""Structure of kb/ or raw/, and the relative paths it holds. No content."""
top = root / name
if not top.is_dir():
return {"present": False}, []
files, dirs = [], 0
depth_histogram, longest_rel, longest_abs = {}, 0, 0
counts = {
"abs_path_over_240": 0, "abs_path_over_260": 0, "names_with_space": 0,
"names_non_ascii": 0, "names_windows_forbidden_chars": 0,
"names_trailing_dot_or_space": 0, "names_reserved_device": 0,
"names_not_nfc": 0, "case_collisions": 0,
}
def judge(entry: str) -> None:
if " " in entry:
counts["names_with_space"] += 1
if any(ord(ch) > 127 for ch in entry):
counts["names_non_ascii"] += 1
if WINDOWS_FORBIDDEN.search(entry):
counts["names_windows_forbidden_chars"] += 1
if entry.endswith((" ", ".")):
counts["names_trailing_dot_or_space"] += 1
if WINDOWS_RESERVED.match(entry):
counts["names_reserved_device"] += 1
if unicodedata.normalize("NFC", entry) != entry:
counts["names_not_nfc"] += 1
for current, subdirs, names in os.walk(str(top)):
entries = sorted(subdirs) + sorted(names)
seen = {}
for entry in entries:
seen.setdefault(entry.casefold(), []).append(entry)
judge(entry)
counts["case_collisions"] += sum(1 for group in seen.values() if len(group) > 1)
dirs += len(subdirs)
for entry in names:
absolute = os.path.join(current, entry)
rel = os.path.relpath(absolute, str(root)).replace(os.sep, "/")
files.append(rel)
depth = rel.count("/") + 1
depth_histogram[str(depth)] = depth_histogram.get(str(depth), 0) + 1
longest_rel = max(longest_rel, len(rel))
longest_abs = max(longest_abs, len(os.path.abspath(absolute)))
if len(os.path.abspath(absolute)) > 240:
counts["abs_path_over_240"] += 1
if len(os.path.abspath(absolute)) > 260:
counts["abs_path_over_260"] += 1
summary = {
"present": True, "files": len(files), "directories": dirs,
"depth_histogram": depth_histogram, "longest_relative_path": longest_rel,
"longest_absolute_path": longest_abs,
}
summary.update(counts)
return summary, files
def collect_stack(root: Path, red: Redactor) -> dict:
version = root / "VERSION"
configs = {}
for path in sorted(root.glob(".wikitool-*.json")):
if path.name == ".wikitool-tools.json":
continue
try:
configs[path.name] = red.clean_json(json.loads(read_text(path)))
except ValueError:
configs[path.name] = {"unparseable": True, "bytes": path.stat().st_size}
return {
"VERSION": read_text(version).strip() if version.is_file() else "absent",
"config_files": configs,
"git": _collect_git_state(root),
}
def _collect_git_state(root: Path) -> dict:
quiet = ["git", "-c", "core.quotepath=off"]
inside = run(quiet + ["rev-parse", "--is-inside-work-tree"], root)
if not inside.ok:
return {"repository": False, "error": inside.first_line()}
branch = run(quiet + ["rev-parse", "--abbrev-ref", "HEAD"], root)
status = run(quiet + ["status", "--porcelain", "-z"], root)
log = run(quiet + ["log", "-n", "20", "--name-only", "--format=%x1e%h%x1f%s"], root)
return {
"repository": True,
"branch": branch.stdout.strip() if branch.ok else branch.first_line(),
"status": _parse_status(status.stdout) if status.ok else [status.first_line()],
"log": _parse_log(log.stdout) if log.ok else [log.first_line()],
}
def _parse_status(raw: str) -> list:
entries = raw.split("\0")
lines, i = [], 0
while i < len(entries):
entry = entries[i]
i += 1
if len(entry) < 4:
continue
code, path = entry[:2], entry[3:]
if code[0] in "RC" and i < len(entries):
lines.append("%s %s (from %s)" % (code, path, entries[i]))
i += 1
else:
lines.append("%s %s" % (code, path))
return lines
def _parse_log(raw: str) -> list:
lines = []
for block in raw.split("\x1e"):
if not block.strip():
continue
head, _, rest = block.partition("\n")
sha, _, subject = head.partition("\x1f")
files = [line for line in rest.splitlines() if line.strip()]
if files and all(f.startswith(tuple(d + "/" for d in CONTENT_COMMIT_DIRS)) for f in files):
parts = []
for directory in CONTENT_COMMIT_DIRS:
count = sum(1 for f in files if f.startswith(directory + "/"))
if count:
parts.append("%s %d" % (directory, count))
subject = "<content commit: %s files>" % ", ".join(parts)
lines.append("%s %s" % (sha.strip(), subject.strip()))
return lines
# ---------------------------------------------------------------------- layer 3: wikitool
def launcher_command(root: Path):
tools = root / "tools"
if sys.platform == "win32":
ps1, sh = tools / "wikitool.ps1", tools / "wikitool"
pwsh = shutil.which("pwsh")
if ps1.is_file() and pwsh:
# Bypass: a policy that blocks the script is a finding for `doctor`'s
# execution-policy check, and must not also stop the report from running.
return [pwsh, "-NoProfile", "-ExecutionPolicy", "Bypass", "-File", str(ps1)]
bash = shutil.which("bash")
if sh.is_file() and bash:
return [bash, str(sh)]
return None
launcher = tools / "wikitool"
return [str(launcher)] if launcher.is_file() else None
def resolve_trace_session(explicit):
if explicit:
return explicit, "--session"
for name in (SESSION_ENV, HARNESS_SESSION_ENV):
if os.environ.get(name):
return os.environ[name], name
return None, None
def trace_root(root: Path) -> Path:
override = os.environ.get("WIKI_TRACE_DIR")
return Path(override) if override else root / "reports" / "telemetry"
def collect_trace(root: Path, explicit):
"""The caller's session trace, or a listing of what exists instead."""
session, source = resolve_trace_session(explicit)
base = trace_root(root)
if session:
path = base / session_slug(session) / "trace.jsonl"
if path.is_file():
return {"found": True, "session": session, "source": source,
"text": read_text(path)}
listing = []
if base.is_dir():
for child in sorted(base.iterdir()):
if child.is_dir() and not child.name.startswith("bugreport-"):
trace = child / "trace.jsonl"
listing.append({"session_directory": child.name,
"bytes": trace.stat().st_size if trace.is_file() else 0})
return {"found": False, "session": session, "source": source, "listing": listing}
WIKITOOL_CALLS = (
("version show", ("version", "show"), "inherited"),
("doctor", ("doctor",), "inherited"),
("budget status", ("budget", "status"), "inherited"),
("instructions verify", ("instructions", "verify"), "own"),
("docs verify", ("docs", "verify"), "own"),
)
def run_wikitool(root: Path, launcher, own_session: str) -> dict:
"""Five fixed read commands. `version show` is the startup probe: if it fails,
`wikitool` counts as not started and nothing else is run."""
outputs = []
started = False
reason = None
for label, args, session in WIKITOOL_CALLS:
env = dict(os.environ)
if session == "own":
env[SESSION_ENV] = own_session
result = run(launcher + list(args), root, env=env, timeout=TIMEOUT_SECONDS)
prefix = "%s=%s " % (SESSION_ENV, own_session) if session == "own" else ""
outputs.append({
"label": label,
"file": "wikitool/%s.txt" % label.replace(" ", "-"),
"session": ("%s (set by the collector)" % own_session) if session == "own"
else "inherited from the caller",
"text": (
"command: %s%s\nsession: %s\nexit: %s\n%s\n--- stdout ---\n%s\n--- stderr ---\n%s\n"
% (prefix, " ".join(launcher + list(args)),
"own" if session == "own" else "inherited",
result.returncode if result.returncode is not None else result.note,
("note: " + result.note) if result.note and result.returncode is not None
else "", result.stdout, result.stderr)
),
})
if label == "version show":
if not result.ok:
reason = "version show %s: %s" % (
("exit %s" % result.returncode) if result.returncode is not None
else result.note, result.first_line())
break
started = True
return {"started": started, "reason": reason, "outputs": outputs}
# ------------------------------------------------------------------------------ bundle
class Bundle:
def __init__(self, directory: Path) -> None:
self.directory = directory
self.files = []
def add(self, rel: str, text: str, layer: str, description: str, note: str = "") -> None:
write_text(self.directory / rel, text)
self.files.append((rel, layer, description, note))
def stamp_now() -> str:
return datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
def unique_stamp(out: Path) -> str:
stamp = stamp_now()
candidate, n = stamp, 1
while (out / ("bugreport-" + candidate)).exists() or (out / ("bugreport-%s.zip" % candidate)).exists():
n += 1
candidate = "%s-%d" % (stamp, n)
return candidate
def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str:
lines = ["# Bug report bundle", "",
"Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"),
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "",
STAGE1_NOTICE if args.pseudonymise else PRIVACY_NOTICE, "",
"## Options", "",
"- chronology: %s" % ("supplied" if args.chronology else "not supplied"),
"- transcripts: %d" % len(args.transcript),
"- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"),
] + (["- pseudonymisation (`--pseudonymise`): stage 1 applied, stage 2 not yet applied"]
if args.pseudonymise else []) + [
"- session trace: %s" % (
"excluded (`--no-trace`)" if args.no_trace else
("included" if trace and trace.get("found") else "not found")),
"", "## wikitool", ""]
if wikitool is None:
lines.append("`wikitool` did not start: no launcher found under `tools/`.")
elif wikitool["started"]:
lines.append("`wikitool` started. Five read commands ran; outputs are verbatim.")
else:
lines.append("`wikitool` did not start (%s). The four other commands were not run."
% wikitool["reason"])
if wikitool:
lines += ["", "| Command | Session id |", "|---|---|"]
for out in wikitool["outputs"]:
lines.append("| `%s` | %s |" % (out["label"], out["session"]))
caller, source = resolve_trace_session(args.session)
lines += ["", "Caller session id: %s" % (
"%s (from %s)" % (caller, source) if caller else
"not set; `wikitool` falls back to its parent process id"), ""]
if trace and not trace.get("found") and not args.no_trace:
lines += ["No trace found for the caller's session%s. Existing session directories "
"(name and size only):" % (
" `%s`" % trace["session"] if trace.get("session") else ""), ""]
for item in trace.get("listing", [])[:100]:
lines.append("- `%s` (%d bytes)" % (item["session_directory"], item["bytes"]))
if not trace.get("listing"):
lines.append("- none")
lines.append("")
lines += ["## Files", "", "| File | Layer | Contents | Note |", "|---|---|---|---|"]
for rel, layer, description, note in sorted(bundle.files, key=lambda f: f[0]):
lines.append("| `%s` | %s | %s | %s |" % (rel, layer, description, note))
lines.append("| `MANIFEST.md` | - | this file | |")
if gaps:
lines += ["", "## Gaps", ""] + ["- %s" % gap for gap in gaps]
return "\n".join(lines) + "\n"
def final_pass(bundle_dir: Path, red: Redactor, shield, skip) -> None:
for path in sorted(bundle_dir.rglob("*")):
if not path.is_file():
continue
rel = path.relative_to(bundle_dir).as_posix()
text = read_text(path)
cleaned = red.scrub(text)
if shield is not None and rel not in skip:
cleaned = shield.apply(cleaned)
if cleaned != text:
write_text(path, cleaned)
def make_zip(bundle_dir: Path, archive: Path) -> None:
with zipfile.ZipFile(str(archive), "w", zipfile.ZIP_DEFLATED) as zf:
for path in sorted(bundle_dir.rglob("*")):
if path.is_file():
zf.write(str(path), bundle_dir.name + "/" + path.relative_to(bundle_dir).as_posix())
def parse_args(argv):
parser = argparse.ArgumentParser(
prog="tools/bugreport",
description="Collect a bug-report bundle. Removes secrets, keeps page titles out "
"unless --titles is given, uploads nothing.",
)
parser.add_argument("--chronology", help="the agent's fact-only chronology (Markdown)")
parser.add_argument("--transcript", action="append", default=[],
help="a harness transcript to include (repeatable; only on request)")
group = parser.add_mutually_exclusive_group()
group.add_argument("--session", help="session id whose trace to include")
group.add_argument("--no-trace", action="store_true", help="leave the session trace out")
parser.add_argument("--titles", action="store_true",
help="keep page titles and paths (writes tree-paths.txt)")
parser.add_argument("--pseudonymise", action="store_true",
help="replace known identities by consistent, shape-preserving "
"placeholders (stage 1); the mapping stays beside the bundle")
parser.add_argument("--root", help="checkout root (default: parent of tools/)")
parser.add_argument("--out", help="output directory (default: <root>/reports)")
parser.add_argument("--bundle", help="stage 2: an existing pseudonymised bundle directory")
parser.add_argument("--candidates", help="stage 2: file with one further name per line "
"('#' starts a comment)")
args = parser.parse_args(argv)
if bool(args.bundle) != bool(args.candidates):
parser.error("--bundle and --candidates belong together")
if args.bundle and (args.chronology or args.transcript or args.session or args.no_trace
or args.titles or args.pseudonymise or args.root or args.out):
parser.error("--bundle/--candidates excludes every collection option")
return args
def main(argv=None) -> int:
args = parse_args(argv)
if args.bundle:
return run_stage2(Path(args.bundle), Path(args.candidates))
root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent
out = Path(args.out).resolve() if args.out else root / "reports"
inputs = [("chronology", args.chronology)] + [("transcript", t) for t in args.transcript]
for label, value in inputs:
if value and not Path(value).is_file():
sys.stderr.write("bugreport: %s file not found: %s\n" % (label, value))
return 1
try:
out.mkdir(parents=True, exist_ok=True)
stamp = unique_stamp(out)
bundle_dir = out / ("bugreport-" + stamp)
bundle_dir.mkdir()
return _collect(args, root, out, stamp, bundle_dir)
except OSError as exc:
sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc)
return 1
def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int:
red = Redactor()
bundle = Bundle(bundle_dir)
gaps = []
trace = None
if not args.no_trace:
trace = guarded(collect_trace, root, args.session)
if trace.get("found"):
bundle.add("trace.jsonl", trace["text"], "3", "session trace of the caller",
MAY_CONTAIN_CONTENT)
bundle.add("environment.json", dump_json(guarded(collect_environment, root, red)), "1",
"OS, Python, shells, harness, PATH, environment variables (names; values for a "
"fixed list), git configuration, venv, line endings, Windows details")
stack = guarded(collect_stack, root, red)
bundle.add("stack.json", dump_json(stack), "2",
"VERSION, `.wikitool-*.json` (secrets removed), git status and log")
content_paths = []
tree = {}
for name in CONTENT_DIRS:
summary, files = _safe_scan(root, name)
tree[name] = summary
content_paths.extend(files)
bundle.add("tree-structure.json", dump_json(tree), "2",
"kb/ and raw/: counts, depth, path lengths, name problems - no content")
if args.titles:
bundle.add("tree-paths.txt", "\n".join(sorted(content_paths)) + "\n", "2",
"all relative paths under kb/ and raw/", "contains page titles")
wikitool = None
launcher = launcher_command(root)
if launcher:
wikitool = run_wikitool(root, launcher, "bugreport-" + stamp)
for item in wikitool["outputs"]:
bundle.add(item["file"], item["text"], "3", "verbatim output of `%s`" % item["label"])
else:
gaps.append("no launcher found under tools/, so no wikitool command ran")
if args.chronology:
bundle.add("CHRONOLOGY.md", read_text(Path(args.chronology)), "4",
"the agent's chronology", MAY_CONTAIN_CONTENT)
else:
gaps.append("no chronology was supplied")
used = set()
for source in args.transcript:
name = Path(source).name
target, n = name, 1
while target in used:
n += 1
target = "%d-%s" % (n, name)
used.add(target)
bundle.add("transcripts/" + target, read_text(Path(source)), "4",
"harness transcript", MAY_CONTAIN_CONTENT)
write_text(bundle_dir / "MANIFEST.md",
build_manifest(args, root, stamp, bundle, wikitool, trace, gaps))
shield = None
if not args.titles:
shield = TitleShield(
rel for rel in content_paths if rel.rsplit("/", 1)[-1] not in STACK_NAMES
)
final_pass(bundle_dir, red, shield, {"tree-paths.txt"})
pz = None
if args.pseudonymise:
pz = Pseudonymiser()
for original, kind in collect_identities(root):
pz.add_identity(original, kind)
pz.finish()
pseudonymise_bundle(bundle_dir, pz)
manifest_path = bundle_dir / "MANIFEST.md"
write_text(manifest_path, replace_section(
read_text(manifest_path), "Pseudonyms", pseudonym_table(pz)))
write_text(_sibling(bundle_dir, ".pseudonyms.json"), dump_json(pz.to_json(bundle_dir.name)))
write_text(_sibling(bundle_dir, ".review.txt"), build_review(bundle_dir, pz))
archive = out / ("bugreport-%s.zip" % stamp)
make_zip(bundle_dir, archive)
print("Bundle: %s" % bundle_dir)
print("Archive: %s" % archive)
if pz is not None:
print("Mapping: %s" % _sibling(bundle_dir, ".pseudonyms.json"))
print("Review: %s" % _sibling(bundle_dir, ".review.txt"))
print("Both hold originals and stay on this machine: never share them.")
print("Stage 2: tools/bugreport --bundle %s --candidates <file>" % bundle_dir)
print()
print(STAGE1_NOTICE if pz is not None else PRIVACY_NOTICE)
return 0
def _sibling(bundle_dir: Path, suffix: str) -> Path:
return bundle_dir.with_name(bundle_dir.name + suffix)
def parse_candidates(text: str):
names = []
for line in text.splitlines():
line = re.sub(r"(^|\s)#.*$", "", line).strip()
if line and line not in names:
names.append(line)
return names
def run_stage2(bundle_dir: Path, candidates: Path) -> int:
bundle_dir = bundle_dir.resolve()
mapping = _sibling(bundle_dir, ".pseudonyms.json")
problem = None
if not bundle_dir.is_dir():
problem = "bundle directory not found: %s" % bundle_dir
elif not candidates.is_file():
problem = "candidates file not found: %s" % candidates
elif not mapping.is_file():
problem = ("no mapping beside the bundle (%s): stage 2 needs the salt stage 1 used, so "
"collect the report again with --pseudonymise" % mapping.name)
if problem:
sys.stderr.write("bugreport: %s\n" % problem)
return 1
try:
pz = Pseudonymiser.from_json(json.loads(read_text(mapping)))
except (ValueError, KeyError) as exc:
sys.stderr.write("bugreport: the mapping cannot be read: %s\n" % exc)
return 1
files = text_files(bundle_dir)
corpus = "\n".join(read_text(path) for path in files)
applied, skipped = [], []
placeholders = pz.placeholder_words()
for name in parse_candidates(read_text(candidates)):
words = _words(name)
if len(name) < MIN_IDENTITY or not words:
skipped.append((name, "under %d characters" % MIN_IDENTITY))
elif all(w in VOCABULARY or w in pz.keep for w in words):
skipped.append((name, "system vocabulary"))
elif all(w in placeholders for w in words):
skipped.append((name, "already a placeholder"))
elif pz.find(corpus, name) == 0:
skipped.append((name, "not found in the bundle"))
else:
applied.append(name)
try:
for name in applied:
pz.add_identity(name, "stage-2 candidate", stage=2)
pz.stage = 2
pseudonymise_bundle(bundle_dir, pz, stage=2)
manifest_path = bundle_dir / "MANIFEST.md"
manifest = read_text(manifest_path)
manifest = replace_section(manifest, "Privacy", RESIDUAL_NOTICE)
manifest = replace_section(manifest, "Pseudonyms", pseudonym_table(pz))
manifest = re.sub(r"(?m)^- pseudonymisation \(`--pseudonymise`\): .*$",
"- pseudonymisation (`--pseudonymise`): stage 1 and stage 2 applied",
manifest)
write_text(manifest_path, manifest)
write_text(mapping, dump_json(pz.to_json(bundle_dir.name)))
archive = bundle_dir.with_name(bundle_dir.name + ".zip")
temporary = archive.with_name(archive.name + ".tmp")
make_zip(bundle_dir, temporary)
os.replace(str(temporary), str(archive))
except OSError as exc:
sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc)
return 1
print("Bundle: %s" % bundle_dir)
print("Archive: %s" % archive)
print("Mapping: %s (holds originals; delete it after the last stage 2 run)" % mapping)
for name in applied:
print("applied: %s" % name)
for name, reason in skipped:
print("not applied: %s - %s" % (name, reason))
print()
print(RESIDUAL_NOTICE)
return 0
def _safe_scan(root: Path, name: str):
try:
return scan_tree(root, name)
except Exception as exc: # noqa: BLE001
return {"collector_error": "%s: %s" % (type(exc).__name__, exc)}, []
if __name__ == "__main__":
sys.exit(main())