feat: bug-report collector pseudonymises identities in two stages, opt-in via --pseudonymise (#158)
CI / verify (push) Successful in 2m5s
Release / release (push) Successful in 39s

Files changed:
- .gitea/workflows/ci.yml
- CHANGES.md
- INSTALL.md
- VERSION
- instructions/bug-report.md
- reports/CONTRACT.md
- tools/README.md
- tools/bugreport.py
- tools/chemenu/tests/test_bugreport.py
This commit is contained in:
torben committed 2026-09-30 19:03:59 +02:00
1 parent 0899c670fe
commit 76d67e45ba
9 files changed
+850 -15

No files matched your search

+491 -5
View File
@@ -8,12 +8,16 @@ Runs on the base Python with the standard library only, and imports nothing from
python tools/bugreport.py [--chronology FILE] [--transcript FILE]...
[--session ID | --no-trace] [--titles]
[--root DIR] [--out DIR]
[--pseudonymise] [--root DIR] [--out DIR]
python tools/bugreport.py --bundle DIR --candidates FILE
The bundle is `<out>/bugreport-<UTC stamp>/` plus a zip beside it, in four
layers: the environment, the stack, what `wikitool` prints (if it starts), and
the agent's chronology. Secrets are always removed. Page titles are kept out of
everything this script generates unless `--titles` is given. Nothing is uploaded.
everything this script generates unless `--titles` is given. With `--pseudonymise`
known identities are replaced by consistent, shape-preserving placeholders (stage 1);
`--bundle`/`--candidates` applies the names a model found beyond them (stage 2). The
mapping stays beside the bundle, never in it. Nothing is uploaded.
Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script
opens no gate. See instructions/bug-report.md.
@@ -22,11 +26,16 @@ from __future__ import annotations
import argparse
import datetime
import getpass
import hashlib
import hmac
import json
import os
import platform
import re
import secrets
import shutil
import socket
import subprocess
import sys
import unicodedata
@@ -78,6 +87,43 @@ PRIVACY_NOTICE = (
"yourself; neither this tool nor the agent uploads anything."
)
STAGE1_NOTICE = (
"Pseudonymisation: stage 1 applied, stage 2 not yet applied. Identities this script could read "
"from the machine (user, host, home and repository paths, git identity, remotes) are replaced by "
"consistent placeholders that keep length, spaces, hyphens, character classes, separators and depth. "
"Names it cannot know - people, companies, customers, internal hosts, projects - may remain, above "
"all in the trace, the chronology and transcripts. Secrets are removed. Read the bundle before you "
"share it. Choose the channel yourself; neither this tool nor the agent uploads anything."
)
RESIDUAL_NOTICE = (
"Pseudonymisation: stage 1 and stage 2 applied. Stage 2 is the judgement of a model: it can miss "
"names, companies, hosts and projects, above all in the free text of a trace or transcript over "
"100 KB, which the model did not read in full. The forms - lengths, character classes, separators "
"and depth - are kept on purpose, so a rare name can still be recognisable by its shape. Read the "
"bundle before you share it. Choose the channel yourself; neither this tool nor the agent uploads "
"anything."
)
PUBLIC_ORIGIN = "gitea.nehmer.net/torben/chemenu" # same origin as chemenu/version.py; this script imports nothing from chemenu
MIN_IDENTITY = 3
NON_ASCII_LOWER = "äöüéèêàáâçñõøåæ"
WORD = re.compile(r"[^\W_]+")
VOCABULARY = frozenset(
w.lower() for w in (
"Windows System32 SysWOW64 Program Files ProgramData Users Public AppData Local LocalLow "
"Roaming Temp Microsoft WindowsApps Programs Documents Desktop Downloads OneDrive "
"DESKTOP LAPTOP "
"home usr local bin opt etc var tmp mnt src Library Applications "
"Python Git Scripts venv chocolatey "
"chemenu wikitool tools reports kb raw bugreport "
"GmbH AG KG SE Inc Ltd LLC "
"removed collected title depth len space "
"json jsonl md txt py ps1 zip cfg template"
).split()
)
WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE)
WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]')
WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]")
@@ -215,6 +261,328 @@ def _title_flags(title: str) -> str:
return " ".join(flags)
# ------------------------------------------------------------------------ pseudonymisation
def _words(text: str):
return [m.group(0).lower() for m in WORD.finditer(text)]
class Pseudonymiser:
"""Consistent, shape-preserving replacement of identities, word by word.
A placeholder hangs on the word, not on the identity: "Beispiel" is replaced the
same way inside "OneDrive - Beispiel GmbH" (stage 1) and inside the candidate
"Beispiel GmbH" (stage 2), so the two stages agree without coordinating.
"""
def __init__(self, salt=None) -> None:
self.salt = salt or secrets.token_hex(16)
self.words = {}
self.keep = set()
self.entries = []
self.skipped = 0
self.stage = 1
self._index = {}
self._taken = set(VOCABULARY)
self._regex = {}
self._forms = {}
# -- persistence
def to_json(self, bundle_name: str) -> dict:
return {
"schema": 1, "bundle": bundle_name, "stage": self.stage, "salt": self.salt,
"keep": sorted(self.keep), "skipped": self.skipped, "words": self.words,
"entries": self.entries,
}
@classmethod
def from_json(cls, data: dict) -> "Pseudonymiser":
pz = cls(data["salt"])
pz.stage = data.get("stage", 1)
pz.keep = set(data.get("keep", []))
pz.skipped = data.get("skipped", 0)
pz.words = dict(data.get("words", {}))
pz.entries = list(data.get("entries", []))
for key, pseudo in pz.words.items():
pz._taken.add(key)
pz._taken.add(pseudo)
for index, entry in enumerate(pz.entries):
pz._index[entry["original"].lower()] = index
pz._taken.update(_words(entry["original"]))
return pz
# -- placeholders
def _stream(self, key: str, counter: int):
block = 0
salt = bytes.fromhex(self.salt)
while True:
message = "%s\x00%d\x00%d" % (key, counter, block)
for byte in hmac.new(salt, message.encode("utf-8"), hashlib.sha256).digest():
yield byte
block += 1
def pseudo_word(self, word: str) -> str:
key = word.lower()
if key in VOCABULARY or key in self.keep:
return word
base = self.words.get(key)
if base is None:
counter = 0
while True:
stream = self._stream(key, counter)
chars = []
for char in word:
byte = next(stream)
if char.isdigit():
chars.append(str(byte % 10))
elif char.isascii():
chars.append(chr(ord("a") + byte % 26))
else:
chars.append(NON_ASCII_LOWER[byte % len(NON_ASCII_LOWER)])
base = "".join(chars)
if (base != key and base not in self._taken) or counter >= 500:
break
counter += 1
self.words[key] = base
self._taken.add(base)
if len(base) != len(word):
base = (base * (len(word) // max(len(base), 1) + 1))[:len(word)]
return "".join(
b.upper() if orig.isupper() and not b.isdigit() else b for orig, b in zip(word, base)
)
def pseudo_text(self, text: str) -> str:
return WORD.sub(lambda m: self.pseudo_word(m.group(0)), text)
# -- identities
def add_identity(self, original: str, kind: str, stage: int = 1) -> bool:
original = original.strip()
if not original or original.lower() in self._index:
return False
words = _words(original)
if len(original) < MIN_IDENTITY or not words:
self.skipped += 1
return False
if kind in ("host", "domain", "remote") and "." in original:
label = _words(original)[-1]
if not label.isdigit():
self.keep.add(label)
if all(w in VOCABULARY or w in self.keep for w in words):
self.skipped += 1
return False
self._taken.update(words)
self._index[original.lower()] = len(self.entries)
self.entries.append({"original": original, "placeholder": None, "kind": kind,
"stage": stage, "count": 0})
self._regex = {}
return True
def finish(self) -> None:
for entry in self.entries:
if entry["placeholder"] is None:
entry["placeholder"] = self.pseudo_text(entry["original"])
# -- applying
def _build(self, stage) -> None:
self.finish()
forms = {PUBLIC_ORIGIN.lower(): (None, False)}
for entry in self.entries:
if stage is not None and entry["stage"] != stage:
continue
raw = entry["original"]
forms.setdefault(raw.lower(), (entry, False))
escaped = json.dumps(raw, ensure_ascii=True)[1:-1]
if escaped.lower() != raw.lower():
forms.setdefault(escaped.lower(), (entry, True))
names = sorted(forms, key=len, reverse=True)
self._regex[stage] = (forms, re.compile(
r"(?<![^\W_])(?:%s)(?![^\W_])" % "|".join(re.escape(n) for n in names), re.IGNORECASE
))
def apply(self, text: str, stage=None) -> str:
if not any(stage is None or e["stage"] == stage for e in self.entries):
return text
if stage not in self._regex:
self._build(stage)
forms, regex = self._regex[stage]
def repl(match):
found = forms.get(match.group(0).lower())
if found is None or found[0] is None:
return match.group(0)
entry, escaped = found
if escaped:
try:
decoded = json.loads('"%s"' % match.group(0))
except ValueError:
return match.group(0)
entry["count"] += 1
return json.dumps(self.pseudo_text(decoded), ensure_ascii=True)[1:-1]
entry["count"] += 1
return self.pseudo_text(match.group(0))
return regex.sub(repl, text)
def find(self, text: str, original: str) -> int:
raw = re.escape(original)
escaped = re.escape(json.dumps(original, ensure_ascii=True)[1:-1])
pattern = re.compile(r"(?<![^\W_])(?:%s|%s)(?![^\W_])" % (raw, escaped), re.IGNORECASE)
return len(pattern.findall(text))
def placeholder_words(self) -> set:
return set(self.words.values())
def _path_parts(value: str):
return [part for part in re.split(r"[\\/]+", value) if part]
def _remote_identities(url: str):
"""(original, kind) pairs for one remote URL. The stack's public origin is exempt."""
url = url.strip()
if not url or PUBLIC_ORIGIN in url.lower().replace("\\", "/"):
return []
found = []
scheme = re.match(r"^[A-Za-z][A-Za-z0-9+.\-]*://", url)
scp = re.match(r"^(?:[^@/\\:\s]+@)?([^/\\:\s]{2,}):(?![/\\])(.*)$", url)
if scheme:
rest = url[scheme.end():]
netloc, _, path = rest.partition("/")
host = netloc.rsplit("@", 1)[-1].split(":")[0]
found.append((host, "remote"))
parts = _path_parts(path)
elif scp:
found.append((scp.group(1), "remote"))
parts = _path_parts(scp.group(2))
else:
parts = _path_parts(url)
found.extend((part, "remote") for part in parts)
return found
def collect_identities(root: Path):
"""Identities readable from this machine, in the order they are registered."""
found = []
user = set()
try:
user.add(getpass.getuser())
except Exception: # noqa: BLE001 - no account name is not a failure
pass
for name in ("USER", "USERNAME", "LOGNAME"):
if os.environ.get(name):
user.add(os.environ[name])
found.extend((value, "user name") for value in sorted(user))
if os.environ.get("USERDOMAIN"):
found.append((os.environ["USERDOMAIN"], "domain"))
if os.environ.get("COMPUTERNAME"):
found.append((os.environ["COMPUTERNAME"], "host"))
try:
found.append((socket.gethostname(), "host"))
except OSError:
pass
found.extend((part, "home path") for part in _path_parts(os.path.expanduser("~")))
found.extend((part, "repo path") for part in _path_parts(str(root)))
for key in ("user.name", "user.email"):
result = run(["git", "config", "--get", key], root)
value = result.stdout.strip() if result.ok else ""
if value:
found.append((value, "git identity"))
if key == "user.email" and "@" in value:
local, _, domain = value.rpartition("@")
found.append((local, "git identity"))
found.append((domain, "domain"))
remotes = run(["git", "remote", "-v"], root)
if remotes.ok:
for line in remotes.stdout.splitlines():
fields = line.split("\t", 1)
if len(fields) == 2:
found.extend(_remote_identities(re.sub(r"\s+\((?:fetch|push)\)\s*$", "", fields[1])))
return found
def text_files(bundle_dir: Path):
return [p for p in sorted(bundle_dir.rglob("*")) if p.is_file()]
def pseudonymise_bundle(bundle_dir: Path, pz: Pseudonymiser, stage=None) -> None:
for path in text_files(bundle_dir):
text = read_text(path)
changed = pz.apply(text, stage)
if changed != text:
write_text(path, changed)
EMAIL_RE = re.compile(r"[\w.+\-]+@[\w\-]+(?:\.[\w\-]+)+")
URL_RE = re.compile(r"[A-Za-z][A-Za-z0-9+.\-]*://[^\s\"'<>)\]]+")
PATH_RE = re.compile(r"(?:[A-Za-z]:)?(?:[\\/]+[^\\/\s\"'<>|:*?,;()\[\]{}]+){2,}")
def build_review(bundle_dir: Path, pz: Pseudonymiser) -> str:
"""What stage 1 leaves structured and unmasked, for the model that does stage 2."""
found = {}
def note(kind: str, value: str, rel: str) -> None:
words = _words(value)
if len(value) < MIN_IDENTITY or not words:
return
if all(w in VOCABULARY or w in pz.keep or w in pz.placeholder_words() for w in words):
return
entry = found.setdefault((kind, value), {"count": 0, "files": []})
entry["count"] += 1
if rel not in entry["files"]:
entry["files"].append(rel)
for path in text_files(bundle_dir):
rel = path.relative_to(bundle_dir).as_posix()
text = read_text(path)
for match in EMAIL_RE.finditer(text):
note("mail address", match.group(0), rel)
for match in URL_RE.finditer(text):
rest = match.group(0).split("://", 1)[1]
netloc, _, tail = rest.partition("/")
note("host", netloc.rsplit("@", 1)[-1], rel)
for part in _path_parts(tail):
note("url segment", part, rel)
for match in PATH_RE.finditer(text):
for part in _path_parts(match.group(0)):
note("path component", part, rel)
lines = [
"# Review list for stage 2 - local only, never part of the bundle. It holds what stage 1",
"# left in the bundle, which can still be a real name. One line: kind, count, value, files.",
]
for (kind, value), info in sorted(found.items(), key=lambda kv: (-kv[1]["count"], kv[0])):
lines.append("%s\t%d\t%s\t%s" % (kind, info["count"], value, ", ".join(info["files"][:5])))
return "\n".join(lines) + "\n"
def pseudonym_table(pz: Pseudonymiser) -> str:
lines = ["| Placeholder | Kind | Stage | Occurrences |", "|---|---|---|---|"]
for entry in pz.entries:
placeholder = entry["placeholder"].replace("|", "\\|")
lines.append("| `%s` | %s | %d | %d |" % (
placeholder, entry["kind"], entry["stage"], entry["count"]))
lines += ["", "%d identities were left unchanged (under %d characters, or system vocabulary); "
"their values are not listed." % (pz.skipped, MIN_IDENTITY)]
return "\n".join(lines)
def replace_section(manifest: str, heading: str, body: str) -> str:
sections = re.split(r"(?m)^(?=## )", manifest)
block = "## %s\n\n%s\n" % (heading, body)
for index, section in enumerate(sections):
if section.startswith("## %s\n" % heading):
sections[index] = block
break
else:
sections.append(block)
return "\n\n".join(part.strip("\n") for part in sections if part.strip()) + "\n"
# ---------------------------------------------------------------------------- helpers
@@ -751,11 +1119,14 @@ def unique_stamp(out: Path) -> str:
def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str:
lines = ["# Bug report bundle", "",
"Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"),
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", PRIVACY_NOTICE, "",
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "",
STAGE1_NOTICE if args.pseudonymise else PRIVACY_NOTICE, "",
"## Options", "",
"- chronology: %s" % ("supplied" if args.chronology else "not supplied"),
"- transcripts: %d" % len(args.transcript),
"- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"),
] + (["- pseudonymisation (`--pseudonymise`): stage 1 applied, stage 2 not yet applied"]
if args.pseudonymise else []) + [
"- session trace: %s" % (
"excluded (`--no-trace`)" if args.no_trace else
("included" if trace and trace.get("found") else "not found")),
@@ -827,13 +1198,27 @@ def parse_args(argv):
group.add_argument("--no-trace", action="store_true", help="leave the session trace out")
parser.add_argument("--titles", action="store_true",
help="keep page titles and paths (writes tree-paths.txt)")
parser.add_argument("--pseudonymise", action="store_true",
help="replace known identities by consistent, shape-preserving "
"placeholders (stage 1); the mapping stays beside the bundle")
parser.add_argument("--root", help="checkout root (default: parent of tools/)")
parser.add_argument("--out", help="output directory (default: <root>/reports)")
return parser.parse_args(argv)
parser.add_argument("--bundle", help="stage 2: an existing pseudonymised bundle directory")
parser.add_argument("--candidates", help="stage 2: file with one further name per line "
"('#' starts a comment)")
args = parser.parse_args(argv)
if bool(args.bundle) != bool(args.candidates):
parser.error("--bundle and --candidates belong together")
if args.bundle and (args.chronology or args.transcript or args.session or args.no_trace
or args.titles or args.pseudonymise or args.root or args.out):
parser.error("--bundle/--candidates excludes every collection option")
return args
def main(argv=None) -> int:
args = parse_args(argv)
if args.bundle:
return run_stage2(Path(args.bundle), Path(args.candidates))
root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent
out = Path(args.out).resolve() if args.out else root / "reports"
@@ -921,13 +1306,114 @@ def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int:
)
final_pass(bundle_dir, red, shield, {"tree-paths.txt"})
pz = None
if args.pseudonymise:
pz = Pseudonymiser()
for original, kind in collect_identities(root):
pz.add_identity(original, kind)
pz.finish()
pseudonymise_bundle(bundle_dir, pz)
manifest_path = bundle_dir / "MANIFEST.md"
write_text(manifest_path, replace_section(
read_text(manifest_path), "Pseudonyms", pseudonym_table(pz)))
write_text(_sibling(bundle_dir, ".pseudonyms.json"), dump_json(pz.to_json(bundle_dir.name)))
write_text(_sibling(bundle_dir, ".review.txt"), build_review(bundle_dir, pz))
archive = out / ("bugreport-%s.zip" % stamp)
make_zip(bundle_dir, archive)
print("Bundle: %s" % bundle_dir)
print("Archive: %s" % archive)
if pz is not None:
print("Mapping: %s" % _sibling(bundle_dir, ".pseudonyms.json"))
print("Review: %s" % _sibling(bundle_dir, ".review.txt"))
print("Both hold originals and stay on this machine: never share them.")
print("Stage 2: python3 tools/bugreport.py --bundle %s --candidates <file>" % bundle_dir)
print()
print(PRIVACY_NOTICE)
print(STAGE1_NOTICE if pz is not None else PRIVACY_NOTICE)
return 0
def _sibling(bundle_dir: Path, suffix: str) -> Path:
return bundle_dir.with_name(bundle_dir.name + suffix)
def parse_candidates(text: str):
names = []
for line in text.splitlines():
line = re.sub(r"(^|\s)#.*$", "", line).strip()
if line and line not in names:
names.append(line)
return names
def run_stage2(bundle_dir: Path, candidates: Path) -> int:
bundle_dir = bundle_dir.resolve()
mapping = _sibling(bundle_dir, ".pseudonyms.json")
problem = None
if not bundle_dir.is_dir():
problem = "bundle directory not found: %s" % bundle_dir
elif not candidates.is_file():
problem = "candidates file not found: %s" % candidates
elif not mapping.is_file():
problem = ("no mapping beside the bundle (%s): stage 2 needs the salt stage 1 used, so "
"collect the report again with --pseudonymise" % mapping.name)
if problem:
sys.stderr.write("bugreport: %s\n" % problem)
return 1
try:
pz = Pseudonymiser.from_json(json.loads(read_text(mapping)))
except (ValueError, KeyError) as exc:
sys.stderr.write("bugreport: the mapping cannot be read: %s\n" % exc)
return 1
files = text_files(bundle_dir)
corpus = "\n".join(read_text(path) for path in files)
applied, skipped = [], []
placeholders = pz.placeholder_words()
for name in parse_candidates(read_text(candidates)):
words = _words(name)
if len(name) < MIN_IDENTITY or not words:
skipped.append((name, "under %d characters" % MIN_IDENTITY))
elif all(w in VOCABULARY or w in pz.keep for w in words):
skipped.append((name, "system vocabulary"))
elif all(w in placeholders for w in words):
skipped.append((name, "already a placeholder"))
elif pz.find(corpus, name) == 0:
skipped.append((name, "not found in the bundle"))
else:
applied.append(name)
try:
for name in applied:
pz.add_identity(name, "stage-2 candidate", stage=2)
pz.stage = 2
pseudonymise_bundle(bundle_dir, pz, stage=2)
manifest_path = bundle_dir / "MANIFEST.md"
manifest = read_text(manifest_path)
manifest = replace_section(manifest, "Privacy", RESIDUAL_NOTICE)
manifest = replace_section(manifest, "Pseudonyms", pseudonym_table(pz))
manifest = re.sub(r"(?m)^- pseudonymisation \(`--pseudonymise`\): .*$",
"- pseudonymisation (`--pseudonymise`): stage 1 and stage 2 applied",
manifest)
write_text(manifest_path, manifest)
write_text(mapping, dump_json(pz.to_json(bundle_dir.name)))
archive = bundle_dir.with_name(bundle_dir.name + ".zip")
temporary = archive.with_name(archive.name + ".tmp")
make_zip(bundle_dir, temporary)
os.replace(str(temporary), str(archive))
except OSError as exc:
sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc)
return 1
print("Bundle: %s" % bundle_dir)
print("Archive: %s" % archive)
print("Mapping: %s (holds originals; delete it after the last stage 2 run)" % mapping)
for name in applied:
print("applied: %s" % name)
for name, reason in skipped:
print("not applied: %s - %s" % (name, reason))
print()
print(RESIDUAL_NOTICE)
return 0