feat: bug-report collector pseudonymises identities in two stages, opt-in via --pseudonymise (#158)
CI / verify (push) Successful in 2m5s
Release / release (push) Successful in 39s

Files changed:
- .gitea/workflows/ci.yml
- CHANGES.md
- INSTALL.md
- VERSION
- instructions/bug-report.md
- reports/CONTRACT.md
- tools/README.md
- tools/bugreport.py
- tools/chemenu/tests/test_bugreport.py
This commit is contained in:
torben committed 2026-09-30 19:03:59 +02:00
1 parent 0899c670fe
commit 76d67e45ba
9 files changed
+850 -15

No files matched your search

+5 -1
View File
@@ -64,7 +64,11 @@ tools/
from it: it is the bug-report collector ([../instructions/bug-report.md](../instructions/bug-report.md)),
and the case it exists for is a checkout where `wikitool` does not start. It therefore uses the
standard library only, keeps to Python 3.8 syntax so that an old interpreter can still run it, and
exits 0 or 1 - never 42, since it opens no gate. `dist export` ships it with the rest of `tools/`;
exits 0 or 1 - never 42, since it opens no gate. It has a second mode: `--pseudonymise` replaces the
identities it can read from the machine (stage 1), and `--bundle`/`--candidates` applies names a
model found in the result (stage 2), both word by word so that length, separators and depth
survive. The mapping, the review list and the candidate file stay beside the bundle directory.
`dist export` ships it with the rest of `tools/`;
its tests (`tests/test_bugreport.py`) run it on a bare interpreter (`-I -S`) to keep that promise.
**Two consumers, one core.** The CLI is not the only caller any more. The cores
+491 -5
View File
@@ -8,12 +8,16 @@ Runs on the base Python with the standard library only, and imports nothing from
python tools/bugreport.py [--chronology FILE] [--transcript FILE]...
[--session ID | --no-trace] [--titles]
[--root DIR] [--out DIR]
[--pseudonymise] [--root DIR] [--out DIR]
python tools/bugreport.py --bundle DIR --candidates FILE
The bundle is `<out>/bugreport-<UTC stamp>/` plus a zip beside it, in four
layers: the environment, the stack, what `wikitool` prints (if it starts), and
the agent's chronology. Secrets are always removed. Page titles are kept out of
everything this script generates unless `--titles` is given. Nothing is uploaded.
everything this script generates unless `--titles` is given. With `--pseudonymise`
known identities are replaced by consistent, shape-preserving placeholders (stage 1);
`--bundle`/`--candidates` applies the names a model found beyond them (stage 2). The
mapping stays beside the bundle, never in it. Nothing is uploaded.
Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script
opens no gate. See instructions/bug-report.md.
@@ -22,11 +26,16 @@ from __future__ import annotations
import argparse
import datetime
import getpass
import hashlib
import hmac
import json
import os
import platform
import re
import secrets
import shutil
import socket
import subprocess
import sys
import unicodedata
@@ -78,6 +87,43 @@ PRIVACY_NOTICE = (
"yourself; neither this tool nor the agent uploads anything."
)
STAGE1_NOTICE = (
"Pseudonymisation: stage 1 applied, stage 2 not yet applied. Identities this script could read "
"from the machine (user, host, home and repository paths, git identity, remotes) are replaced by "
"consistent placeholders that keep length, spaces, hyphens, character classes, separators and depth. "
"Names it cannot know - people, companies, customers, internal hosts, projects - may remain, above "
"all in the trace, the chronology and transcripts. Secrets are removed. Read the bundle before you "
"share it. Choose the channel yourself; neither this tool nor the agent uploads anything."
)
RESIDUAL_NOTICE = (
"Pseudonymisation: stage 1 and stage 2 applied. Stage 2 is the judgement of a model: it can miss "
"names, companies, hosts and projects, above all in the free text of a trace or transcript over "
"100 KB, which the model did not read in full. The forms - lengths, character classes, separators "
"and depth - are kept on purpose, so a rare name can still be recognisable by its shape. Read the "
"bundle before you share it. Choose the channel yourself; neither this tool nor the agent uploads "
"anything."
)
PUBLIC_ORIGIN = "gitea.nehmer.net/torben/chemenu" # same origin as chemenu/version.py; this script imports nothing from chemenu
MIN_IDENTITY = 3
NON_ASCII_LOWER = "äöüéèêàáâçñõøåæ"
WORD = re.compile(r"[^\W_]+")
VOCABULARY = frozenset(
w.lower() for w in (
"Windows System32 SysWOW64 Program Files ProgramData Users Public AppData Local LocalLow "
"Roaming Temp Microsoft WindowsApps Programs Documents Desktop Downloads OneDrive "
"DESKTOP LAPTOP "
"home usr local bin opt etc var tmp mnt src Library Applications "
"Python Git Scripts venv chocolatey "
"chemenu wikitool tools reports kb raw bugreport "
"GmbH AG KG SE Inc Ltd LLC "
"removed collected title depth len space "
"json jsonl md txt py ps1 zip cfg template"
).split()
)
WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE)
WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]')
WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]")
@@ -215,6 +261,328 @@ def _title_flags(title: str) -> str:
return " ".join(flags)
# ------------------------------------------------------------------------ pseudonymisation
def _words(text: str):
return [m.group(0).lower() for m in WORD.finditer(text)]
class Pseudonymiser:
"""Consistent, shape-preserving replacement of identities, word by word.
A placeholder hangs on the word, not on the identity: "Beispiel" is replaced the
same way inside "OneDrive - Beispiel GmbH" (stage 1) and inside the candidate
"Beispiel GmbH" (stage 2), so the two stages agree without coordinating.
"""
def __init__(self, salt=None) -> None:
self.salt = salt or secrets.token_hex(16)
self.words = {}
self.keep = set()
self.entries = []
self.skipped = 0
self.stage = 1
self._index = {}
self._taken = set(VOCABULARY)
self._regex = {}
self._forms = {}
# -- persistence
def to_json(self, bundle_name: str) -> dict:
return {
"schema": 1, "bundle": bundle_name, "stage": self.stage, "salt": self.salt,
"keep": sorted(self.keep), "skipped": self.skipped, "words": self.words,
"entries": self.entries,
}
@classmethod
def from_json(cls, data: dict) -> "Pseudonymiser":
pz = cls(data["salt"])
pz.stage = data.get("stage", 1)
pz.keep = set(data.get("keep", []))
pz.skipped = data.get("skipped", 0)
pz.words = dict(data.get("words", {}))
pz.entries = list(data.get("entries", []))
for key, pseudo in pz.words.items():
pz._taken.add(key)
pz._taken.add(pseudo)
for index, entry in enumerate(pz.entries):
pz._index[entry["original"].lower()] = index
pz._taken.update(_words(entry["original"]))
return pz
# -- placeholders
def _stream(self, key: str, counter: int):
block = 0
salt = bytes.fromhex(self.salt)
while True:
message = "%s\x00%d\x00%d" % (key, counter, block)
for byte in hmac.new(salt, message.encode("utf-8"), hashlib.sha256).digest():
yield byte
block += 1
def pseudo_word(self, word: str) -> str:
key = word.lower()
if key in VOCABULARY or key in self.keep:
return word
base = self.words.get(key)
if base is None:
counter = 0
while True:
stream = self._stream(key, counter)
chars = []
for char in word:
byte = next(stream)
if char.isdigit():
chars.append(str(byte % 10))
elif char.isascii():
chars.append(chr(ord("a") + byte % 26))
else:
chars.append(NON_ASCII_LOWER[byte % len(NON_ASCII_LOWER)])
base = "".join(chars)
if (base != key and base not in self._taken) or counter >= 500:
break
counter += 1
self.words[key] = base
self._taken.add(base)
if len(base) != len(word):
base = (base * (len(word) // max(len(base), 1) + 1))[:len(word)]
return "".join(
b.upper() if orig.isupper() and not b.isdigit() else b for orig, b in zip(word, base)
)
def pseudo_text(self, text: str) -> str:
return WORD.sub(lambda m: self.pseudo_word(m.group(0)), text)
# -- identities
def add_identity(self, original: str, kind: str, stage: int = 1) -> bool:
original = original.strip()
if not original or original.lower() in self._index:
return False
words = _words(original)
if len(original) < MIN_IDENTITY or not words:
self.skipped += 1
return False
if kind in ("host", "domain", "remote") and "." in original:
label = _words(original)[-1]
if not label.isdigit():
self.keep.add(label)
if all(w in VOCABULARY or w in self.keep for w in words):
self.skipped += 1
return False
self._taken.update(words)
self._index[original.lower()] = len(self.entries)
self.entries.append({"original": original, "placeholder": None, "kind": kind,
"stage": stage, "count": 0})
self._regex = {}
return True
def finish(self) -> None:
for entry in self.entries:
if entry["placeholder"] is None:
entry["placeholder"] = self.pseudo_text(entry["original"])
# -- applying
def _build(self, stage) -> None:
self.finish()
forms = {PUBLIC_ORIGIN.lower(): (None, False)}
for entry in self.entries:
if stage is not None and entry["stage"] != stage:
continue
raw = entry["original"]
forms.setdefault(raw.lower(), (entry, False))
escaped = json.dumps(raw, ensure_ascii=True)[1:-1]
if escaped.lower() != raw.lower():
forms.setdefault(escaped.lower(), (entry, True))
names = sorted(forms, key=len, reverse=True)
self._regex[stage] = (forms, re.compile(
r"(?<![^\W_])(?:%s)(?![^\W_])" % "|".join(re.escape(n) for n in names), re.IGNORECASE
))
def apply(self, text: str, stage=None) -> str:
if not any(stage is None or e["stage"] == stage for e in self.entries):
return text
if stage not in self._regex:
self._build(stage)
forms, regex = self._regex[stage]
def repl(match):
found = forms.get(match.group(0).lower())
if found is None or found[0] is None:
return match.group(0)
entry, escaped = found
if escaped:
try:
decoded = json.loads('"%s"' % match.group(0))
except ValueError:
return match.group(0)
entry["count"] += 1
return json.dumps(self.pseudo_text(decoded), ensure_ascii=True)[1:-1]
entry["count"] += 1
return self.pseudo_text(match.group(0))
return regex.sub(repl, text)
def find(self, text: str, original: str) -> int:
raw = re.escape(original)
escaped = re.escape(json.dumps(original, ensure_ascii=True)[1:-1])
pattern = re.compile(r"(?<![^\W_])(?:%s|%s)(?![^\W_])" % (raw, escaped), re.IGNORECASE)
return len(pattern.findall(text))
def placeholder_words(self) -> set:
return set(self.words.values())
def _path_parts(value: str):
return [part for part in re.split(r"[\\/]+", value) if part]
def _remote_identities(url: str):
"""(original, kind) pairs for one remote URL. The stack's public origin is exempt."""
url = url.strip()
if not url or PUBLIC_ORIGIN in url.lower().replace("\\", "/"):
return []
found = []
scheme = re.match(r"^[A-Za-z][A-Za-z0-9+.\-]*://", url)
scp = re.match(r"^(?:[^@/\\:\s]+@)?([^/\\:\s]{2,}):(?![/\\])(.*)$", url)
if scheme:
rest = url[scheme.end():]
netloc, _, path = rest.partition("/")
host = netloc.rsplit("@", 1)[-1].split(":")[0]
found.append((host, "remote"))
parts = _path_parts(path)
elif scp:
found.append((scp.group(1), "remote"))
parts = _path_parts(scp.group(2))
else:
parts = _path_parts(url)
found.extend((part, "remote") for part in parts)
return found
def collect_identities(root: Path):
"""Identities readable from this machine, in the order they are registered."""
found = []
user = set()
try:
user.add(getpass.getuser())
except Exception: # noqa: BLE001 - no account name is not a failure
pass
for name in ("USER", "USERNAME", "LOGNAME"):
if os.environ.get(name):
user.add(os.environ[name])
found.extend((value, "user name") for value in sorted(user))
if os.environ.get("USERDOMAIN"):
found.append((os.environ["USERDOMAIN"], "domain"))
if os.environ.get("COMPUTERNAME"):
found.append((os.environ["COMPUTERNAME"], "host"))
try:
found.append((socket.gethostname(), "host"))
except OSError:
pass
found.extend((part, "home path") for part in _path_parts(os.path.expanduser("~")))
found.extend((part, "repo path") for part in _path_parts(str(root)))
for key in ("user.name", "user.email"):
result = run(["git", "config", "--get", key], root)
value = result.stdout.strip() if result.ok else ""
if value:
found.append((value, "git identity"))
if key == "user.email" and "@" in value:
local, _, domain = value.rpartition("@")
found.append((local, "git identity"))
found.append((domain, "domain"))
remotes = run(["git", "remote", "-v"], root)
if remotes.ok:
for line in remotes.stdout.splitlines():
fields = line.split("\t", 1)
if len(fields) == 2:
found.extend(_remote_identities(re.sub(r"\s+\((?:fetch|push)\)\s*$", "", fields[1])))
return found
def text_files(bundle_dir: Path):
return [p for p in sorted(bundle_dir.rglob("*")) if p.is_file()]
def pseudonymise_bundle(bundle_dir: Path, pz: Pseudonymiser, stage=None) -> None:
for path in text_files(bundle_dir):
text = read_text(path)
changed = pz.apply(text, stage)
if changed != text:
write_text(path, changed)
EMAIL_RE = re.compile(r"[\w.+\-]+@[\w\-]+(?:\.[\w\-]+)+")
URL_RE = re.compile(r"[A-Za-z][A-Za-z0-9+.\-]*://[^\s\"'<>)\]]+")
PATH_RE = re.compile(r"(?:[A-Za-z]:)?(?:[\\/]+[^\\/\s\"'<>|:*?,;()\[\]{}]+){2,}")
def build_review(bundle_dir: Path, pz: Pseudonymiser) -> str:
"""What stage 1 leaves structured and unmasked, for the model that does stage 2."""
found = {}
def note(kind: str, value: str, rel: str) -> None:
words = _words(value)
if len(value) < MIN_IDENTITY or not words:
return
if all(w in VOCABULARY or w in pz.keep or w in pz.placeholder_words() for w in words):
return
entry = found.setdefault((kind, value), {"count": 0, "files": []})
entry["count"] += 1
if rel not in entry["files"]:
entry["files"].append(rel)
for path in text_files(bundle_dir):
rel = path.relative_to(bundle_dir).as_posix()
text = read_text(path)
for match in EMAIL_RE.finditer(text):
note("mail address", match.group(0), rel)
for match in URL_RE.finditer(text):
rest = match.group(0).split("://", 1)[1]
netloc, _, tail = rest.partition("/")
note("host", netloc.rsplit("@", 1)[-1], rel)
for part in _path_parts(tail):
note("url segment", part, rel)
for match in PATH_RE.finditer(text):
for part in _path_parts(match.group(0)):
note("path component", part, rel)
lines = [
"# Review list for stage 2 - local only, never part of the bundle. It holds what stage 1",
"# left in the bundle, which can still be a real name. One line: kind, count, value, files.",
]
for (kind, value), info in sorted(found.items(), key=lambda kv: (-kv[1]["count"], kv[0])):
lines.append("%s\t%d\t%s\t%s" % (kind, info["count"], value, ", ".join(info["files"][:5])))
return "\n".join(lines) + "\n"
def pseudonym_table(pz: Pseudonymiser) -> str:
lines = ["| Placeholder | Kind | Stage | Occurrences |", "|---|---|---|---|"]
for entry in pz.entries:
placeholder = entry["placeholder"].replace("|", "\\|")
lines.append("| `%s` | %s | %d | %d |" % (
placeholder, entry["kind"], entry["stage"], entry["count"]))
lines += ["", "%d identities were left unchanged (under %d characters, or system vocabulary); "
"their values are not listed." % (pz.skipped, MIN_IDENTITY)]
return "\n".join(lines)
def replace_section(manifest: str, heading: str, body: str) -> str:
sections = re.split(r"(?m)^(?=## )", manifest)
block = "## %s\n\n%s\n" % (heading, body)
for index, section in enumerate(sections):
if section.startswith("## %s\n" % heading):
sections[index] = block
break
else:
sections.append(block)
return "\n\n".join(part.strip("\n") for part in sections if part.strip()) + "\n"
# ---------------------------------------------------------------------------- helpers
@@ -751,11 +1119,14 @@ def unique_stamp(out: Path) -> str:
def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str:
lines = ["# Bug report bundle", "",
"Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"),
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", PRIVACY_NOTICE, "",
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "",
STAGE1_NOTICE if args.pseudonymise else PRIVACY_NOTICE, "",
"## Options", "",
"- chronology: %s" % ("supplied" if args.chronology else "not supplied"),
"- transcripts: %d" % len(args.transcript),
"- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"),
] + (["- pseudonymisation (`--pseudonymise`): stage 1 applied, stage 2 not yet applied"]
if args.pseudonymise else []) + [
"- session trace: %s" % (
"excluded (`--no-trace`)" if args.no_trace else
("included" if trace and trace.get("found") else "not found")),
@@ -827,13 +1198,27 @@ def parse_args(argv):
group.add_argument("--no-trace", action="store_true", help="leave the session trace out")
parser.add_argument("--titles", action="store_true",
help="keep page titles and paths (writes tree-paths.txt)")
parser.add_argument("--pseudonymise", action="store_true",
help="replace known identities by consistent, shape-preserving "
"placeholders (stage 1); the mapping stays beside the bundle")
parser.add_argument("--root", help="checkout root (default: parent of tools/)")
parser.add_argument("--out", help="output directory (default: <root>/reports)")
return parser.parse_args(argv)
parser.add_argument("--bundle", help="stage 2: an existing pseudonymised bundle directory")
parser.add_argument("--candidates", help="stage 2: file with one further name per line "
"('#' starts a comment)")
args = parser.parse_args(argv)
if bool(args.bundle) != bool(args.candidates):
parser.error("--bundle and --candidates belong together")
if args.bundle and (args.chronology or args.transcript or args.session or args.no_trace
or args.titles or args.pseudonymise or args.root or args.out):
parser.error("--bundle/--candidates excludes every collection option")
return args
def main(argv=None) -> int:
args = parse_args(argv)
if args.bundle:
return run_stage2(Path(args.bundle), Path(args.candidates))
root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent
out = Path(args.out).resolve() if args.out else root / "reports"
@@ -921,13 +1306,114 @@ def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int:
)
final_pass(bundle_dir, red, shield, {"tree-paths.txt"})
pz = None
if args.pseudonymise:
pz = Pseudonymiser()
for original, kind in collect_identities(root):
pz.add_identity(original, kind)
pz.finish()
pseudonymise_bundle(bundle_dir, pz)
manifest_path = bundle_dir / "MANIFEST.md"
write_text(manifest_path, replace_section(
read_text(manifest_path), "Pseudonyms", pseudonym_table(pz)))
write_text(_sibling(bundle_dir, ".pseudonyms.json"), dump_json(pz.to_json(bundle_dir.name)))
write_text(_sibling(bundle_dir, ".review.txt"), build_review(bundle_dir, pz))
archive = out / ("bugreport-%s.zip" % stamp)
make_zip(bundle_dir, archive)
print("Bundle: %s" % bundle_dir)
print("Archive: %s" % archive)
if pz is not None:
print("Mapping: %s" % _sibling(bundle_dir, ".pseudonyms.json"))
print("Review: %s" % _sibling(bundle_dir, ".review.txt"))
print("Both hold originals and stay on this machine: never share them.")
print("Stage 2: python3 tools/bugreport.py --bundle %s --candidates <file>" % bundle_dir)
print()
print(PRIVACY_NOTICE)
print(STAGE1_NOTICE if pz is not None else PRIVACY_NOTICE)
return 0
def _sibling(bundle_dir: Path, suffix: str) -> Path:
return bundle_dir.with_name(bundle_dir.name + suffix)
def parse_candidates(text: str):
names = []
for line in text.splitlines():
line = re.sub(r"(^|\s)#.*$", "", line).strip()
if line and line not in names:
names.append(line)
return names
def run_stage2(bundle_dir: Path, candidates: Path) -> int:
bundle_dir = bundle_dir.resolve()
mapping = _sibling(bundle_dir, ".pseudonyms.json")
problem = None
if not bundle_dir.is_dir():
problem = "bundle directory not found: %s" % bundle_dir
elif not candidates.is_file():
problem = "candidates file not found: %s" % candidates
elif not mapping.is_file():
problem = ("no mapping beside the bundle (%s): stage 2 needs the salt stage 1 used, so "
"collect the report again with --pseudonymise" % mapping.name)
if problem:
sys.stderr.write("bugreport: %s\n" % problem)
return 1
try:
pz = Pseudonymiser.from_json(json.loads(read_text(mapping)))
except (ValueError, KeyError) as exc:
sys.stderr.write("bugreport: the mapping cannot be read: %s\n" % exc)
return 1
files = text_files(bundle_dir)
corpus = "\n".join(read_text(path) for path in files)
applied, skipped = [], []
placeholders = pz.placeholder_words()
for name in parse_candidates(read_text(candidates)):
words = _words(name)
if len(name) < MIN_IDENTITY or not words:
skipped.append((name, "under %d characters" % MIN_IDENTITY))
elif all(w in VOCABULARY or w in pz.keep for w in words):
skipped.append((name, "system vocabulary"))
elif all(w in placeholders for w in words):
skipped.append((name, "already a placeholder"))
elif pz.find(corpus, name) == 0:
skipped.append((name, "not found in the bundle"))
else:
applied.append(name)
try:
for name in applied:
pz.add_identity(name, "stage-2 candidate", stage=2)
pz.stage = 2
pseudonymise_bundle(bundle_dir, pz, stage=2)
manifest_path = bundle_dir / "MANIFEST.md"
manifest = read_text(manifest_path)
manifest = replace_section(manifest, "Privacy", RESIDUAL_NOTICE)
manifest = replace_section(manifest, "Pseudonyms", pseudonym_table(pz))
manifest = re.sub(r"(?m)^- pseudonymisation \(`--pseudonymise`\): .*$",
"- pseudonymisation (`--pseudonymise`): stage 1 and stage 2 applied",
manifest)
write_text(manifest_path, manifest)
write_text(mapping, dump_json(pz.to_json(bundle_dir.name)))
archive = bundle_dir.with_name(bundle_dir.name + ".zip")
temporary = archive.with_name(archive.name + ".tmp")
make_zip(bundle_dir, temporary)
os.replace(str(temporary), str(archive))
except OSError as exc:
sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc)
return 1
print("Bundle: %s" % bundle_dir)
print("Archive: %s" % archive)
print("Mapping: %s (holds originals; delete it after the last stage 2 run)" % mapping)
for name in applied:
print("applied: %s" % name)
for name, reason in skipped:
print("not applied: %s - %s" % (name, reason))
print()
print(RESIDUAL_NOTICE)
return 0
+231
View File
@@ -325,3 +325,234 @@ def test_the_collector_ships_with_a_distribution():
from chemenu.commands import dist_cmd
assert "tools/bugreport.py" in dist_cmd.build_plan()
# ------------------------------------------------------------------ pseudonymisation
REMOTE = "C:\\Users\\Max.Muster\\OneDrive - Beispiel GmbH\\x.git"
REPO_ROOT = Path(SCRIPT).resolve().parent.parent
def machine(checkout: Path, monkeypatch, user: str = "Max") -> None:
for name in ("USER", "USERNAME", "LOGNAME"):
monkeypatch.setenv(name, user)
monkeypatch.setenv("COMPUTERNAME", "DESKTOP-Zx91Q")
monkeypatch.setenv("USERDOMAIN", "AB")
monkeypatch.setattr(bugreport.socket, "gethostname", lambda: "Rechner-Delta")
git(checkout, "config", "user.name", "Erika Mustermann")
git(checkout, "config", "user.email", "erika@beispiel-firma.example")
remotes = subprocess.run(["git", "remote"], cwd=checkout, capture_output=True, text=True).stdout
if "corp" not in remotes.split():
git(checkout, "remote", "add", "corp", REMOTE)
def pseudonymised(checkout, tmp_path, monkeypatch, *extra, user="Max", name="out"):
machine(checkout, monkeypatch, user)
out = tmp_path / name
assert bugreport.main(
["--root", str(checkout), "--out", str(out), "--no-trace", "--pseudonymise", *extra]
) == 0
return next(p for p in out.iterdir() if p.is_dir())
def zip_text(bundle: Path) -> str:
archive = bundle.with_name(bundle.name + ".zip")
with zipfile.ZipFile(archive) as zf:
return "\n".join(zf.read(n).decode("utf-8", errors="replace") for n in zf.namelist())
def mapping_of(bundle: Path) -> Path:
return bundle.with_name(bundle.name + ".pseudonyms.json")
def snapshot(bundle: Path) -> dict:
return {p.relative_to(bundle).as_posix(): p.read_bytes() for p in bundle.rglob("*") if p.is_file()}
def test_a_remote_keeps_its_shape_and_loses_its_names(checkout, tmp_path, monkeypatch):
bundle = pseudonymised(checkout, tmp_path, monkeypatch)
assert "Max.Muster" not in all_text(bundle) and "Max.Muster" not in zip_text(bundle)
assert "Beispiel" not in all_text(bundle)
remotes = json.loads((bundle / "environment.json").read_text())["git"]["remotes"]
line = next(r for r in remotes if r.startswith("corp"))
original = f"corp\t{REMOTE} (fetch)".split("\\")
shaped = line.split("\\")
assert len(shaped) == len(original)
assert [len(part) for part in shaped] == [len(part) for part in original]
assert shaped[1] == "Users"
assert shaped[2][3] == "." and shaped[2] != "Max.Muster"
assert shaped[3].startswith("OneDrive - ") and shaped[3].endswith(" GmbH")
assert shaped[4].endswith(".git (fetch)")
def test_one_identity_gets_one_placeholder_in_every_file_and_form(checkout, tmp_path, monkeypatch):
name = "Müller-Lüdenscheidt"
monkeypatch.setenv("PATH", "/home/%s/bin%s%s" % (name, os.pathsep, os.environ["PATH"]))
trace_dir = checkout / "reports" / "telemetry" / "callers-session"
trace_dir.mkdir(parents=True)
(trace_dir / "trace.jsonl").write_text(json.dumps({"p": "/home/" + name}) + "\n")
monkeypatch.setenv("WIKITOOL_SESSION_ID", "callers-session")
chron = tmp_path / "chron.md"
chron.write_text("user %s and %s\n" % (name, name.upper()))
machine(checkout, monkeypatch, user=name)
out = tmp_path / "out"
assert bugreport.main(["--root", str(checkout), "--out", str(out), "--pseudonymise",
"--chronology", str(chron)]) == 0
bundle = next(p for p in out.iterdir() if p.is_dir())
env_entry = json.loads((bundle / "environment.json").read_text())["path_entries"][0]["entry"]
from_env = env_entry.split("/")[2]
from_trace = json.loads((bundle / "trace.jsonl").read_text())["p"].split("/")[2]
lines = (bundle / "CHRONOLOGY.md").read_text().split()
assert from_env == from_trace == lines[1]
assert lines[3] == from_env.upper()
assert from_env != name and len(from_env) == len(name)
for original, shaped in zip(name, from_env):
assert (original == "-") == (shaped == "-")
assert original.isascii() == shaped.isascii()
assert original.isupper() == shaped.isupper()
assert name not in all_text(bundle) and "M\\u00fcller" not in all_text(bundle)
def test_placeholders_keep_length_classes_and_separators_and_are_injective():
pz = bugreport.Pseudonymiser(salt="00" * 16)
assert pz.add_identity("Müller-Lüdenscheidt 42", "user name")
pz.finish()
placeholder = pz.entries[0]["placeholder"]
original = pz.entries[0]["original"]
assert len(placeholder) == len(original)
for a, b in zip(original, placeholder):
assert a.isalpha() == b.isalpha() and a.isdigit() == b.isdigit()
assert a.isascii() == b.isascii() and a.isupper() == b.isupper()
if not a.isalnum():
assert a == b
assert pz.pseudo_text("Users GmbH") == "Users GmbH"
words = ["name%03d" % i for i in range(300)]
shaped = {pz.pseudo_word(w) for w in words}
assert len(shaped) == len(words)
assert not shaped & set(words)
def test_an_identity_matches_whole_words_only(checkout, tmp_path, monkeypatch):
chron = tmp_path / "chron.md"
chron.write_text("Maximum and Max_1 and Max here, Maxine too\n")
bundle = pseudonymised(checkout, tmp_path, monkeypatch, "--chronology", str(chron))
text = (bundle / "CHRONOLOGY.md").read_text()
assert "Maximum" in text and "Maxine" in text
assert text.count("Max") == 2 # "Maximum" and "Maxine" contain none at a word end; Max_1 and Max are replaced
def test_stage_two_applies_candidates_with_the_same_word_and_is_idempotent(
checkout, tmp_path, monkeypatch, capsys
):
chron = tmp_path / "chron.md"
chron.write_text("Firma Beispiel GmbH hat Zugriff. Kunde Acme Corp. Acme Corp zahlt.\n")
bundle = pseudonymised(checkout, tmp_path, monkeypatch, "--chronology", str(chron))
assert "Beispiel GmbH" in (bundle / "CHRONOLOGY.md").read_text()
remotes = json.loads((bundle / "environment.json").read_text())["git"]["remotes"]
stage_one_word = next(r for r in remotes if r.startswith("corp")).split("\\")[3].split()[2]
candidates = tmp_path / "candidates.txt"
candidates.write_text("# from the model\nBeispiel GmbH\nAcme Corp # customer\nab\nUsers\nnowhere at all\n")
capsys.readouterr()
assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 0
printed = capsys.readouterr().out
for needle in ("not applied: ab", "not applied: Users", "not applied: nowhere at all",
"100 KB"):
assert needle in printed
chronology = (bundle / "CHRONOLOGY.md").read_text()
assert "Acme" not in chronology and "Beispiel" not in chronology
assert chronology.startswith("Firma %s GmbH hat" % stage_one_word)
assert "Acme" not in zip_text(bundle) and "Beispiel" not in zip_text(bundle)
first = snapshot(bundle)
mapping_before = mapping_of(bundle).read_bytes()
assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 0
assert snapshot(bundle) == first
assert mapping_of(bundle).read_bytes() == mapping_before
def test_the_mapping_review_and_candidates_stay_beside_the_bundle_and_stage_two_needs_the_mapping(
checkout, tmp_path, monkeypatch, capsys
):
bundle = pseudonymised(checkout, tmp_path, monkeypatch)
review = bundle.with_name(bundle.name + ".review.txt")
assert mapping_of(bundle).is_file() and review.is_file()
assert "Max.Muster" in mapping_of(bundle).read_text()
archive = bundle.with_name(bundle.name + ".zip")
with zipfile.ZipFile(archive) as zf:
assert not [n for n in zf.namelist() if "pseudonyms" in n or "review" in n]
assert not [p for p in bundle.rglob("*") if "pseudonyms" in p.name or "review" in p.name]
candidates = tmp_path / "candidates.txt"
candidates.write_text("Erika\n")
mapping_of(bundle).unlink()
before, archive_before = snapshot(bundle), archive.read_bytes()
assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 1
assert "collect the report again" in capsys.readouterr().err
assert snapshot(bundle) == before and archive.read_bytes() == archive_before
def test_the_manifest_has_three_privacy_versions_and_never_an_original(
checkout, tmp_path, monkeypatch
):
plain = collect(checkout, tmp_path / "plain", "--no-trace")
assert "not pseudonymised" in (plain / "MANIFEST.md").read_text()
assert "## Pseudonyms" not in (plain / "MANIFEST.md").read_text()
bundle = pseudonymised(checkout, tmp_path, monkeypatch)
manifest = (bundle / "MANIFEST.md").read_text()
assert "stage 1 applied, stage 2 not yet applied" in manifest
assert "| Placeholder | Kind | Stage | Occurrences |" in manifest
for original in ("Max.Muster", "Beispiel", "Rechner-Delta", "Mustermann", "beispiel-firma"):
assert original not in manifest
candidates = tmp_path / "candidates.txt"
candidates.write_text("Erika\n")
assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 0
manifest = (bundle / "MANIFEST.md").read_text()
assert "stage 1 and stage 2 applied" in manifest
assert "judgement of a model" in manifest and "100 KB" in manifest
assert "Erika" not in manifest
instruction = (REPO_ROOT / "instructions" / "bug-report.md").read_text()
assert "100 KB" in instruction and "residual" in instruction.lower()
def test_two_bundles_of_one_machine_use_different_placeholders(checkout, tmp_path, monkeypatch):
monkeypatch.setattr(bugreport, "stamp_now", lambda: "20260101T000000Z")
first = pseudonymised(checkout, tmp_path, monkeypatch, name="a")
second = pseudonymised(checkout, tmp_path, monkeypatch, name="b")
def shaped(bundle):
remotes = json.loads((bundle / "environment.json").read_text())["git"]["remotes"]
return next(r for r in remotes if r.startswith("corp")).split("\\")[2]
assert shaped(first) != shaped(second)
def test_the_public_origin_short_and_vocabulary_identities_stay_readable(
checkout, tmp_path, monkeypatch
):
git(checkout, "remote", "add", "pub", "https://gitea.nehmer.net/torben/chemenu.git")
chron = tmp_path / "chron.md"
chron.write_text("torben was here; AB and DESKTOP stay\n")
bundle = pseudonymised(checkout, tmp_path, monkeypatch, "--chronology", str(chron), user="torben")
text = (bundle / "environment.json").read_text()
assert "gitea.nehmer.net/torben/chemenu.git" in text
chronology = (bundle / "CHRONOLOGY.md").read_text()
assert chronology.startswith("torben was") is False
assert "AB and DESKTOP stay" in chronology
manifest = (bundle / "MANIFEST.md").read_text()
assert int(manifest.split(" identities were left unchanged")[0].split()[-1]) >= 2
def test_bundle_and_candidates_exclude_the_collection_options(checkout, tmp_path):
for extra in (["--no-trace"], ["--titles"], ["--pseudonymise"], ["--root", str(checkout)]):
with pytest.raises(SystemExit):
bugreport.main(["--bundle", "b", "--candidates", "c", *extra])
with pytest.raises(SystemExit):
bugreport.main(["--bundle", "b"])
with pytest.raises(SystemExit):
bugreport.main(["--candidates", "c"])
def test_without_the_flag_nothing_but_the_bundle_and_its_archive_is_written(checkout, tmp_path):
collect(checkout, tmp_path, "--no-trace")
assert sorted(p.suffix for p in (tmp_path / "out").iterdir()) == ["", ".zip"]