feat: bug-report collector pseudonymises identities in two stages, opt-in via --pseudonymise (#158)
Files changed: - .gitea/workflows/ci.yml - CHANGES.md - INSTALL.md - VERSION - instructions/bug-report.md - reports/CONTRACT.md - tools/README.md - tools/bugreport.py - tools/chemenu/tests/test_bugreport.py
This commit is contained in:
1 parent
0899c670fe
commit
76d67e45ba
9 files changed
+850
-15
No files matched your search
+491
-5
@@ -8,12 +8,16 @@ Runs on the base Python with the standard library only, and imports nothing from
|
||||
|
||||
python tools/bugreport.py [--chronology FILE] [--transcript FILE]...
|
||||
[--session ID | --no-trace] [--titles]
|
||||
[--root DIR] [--out DIR]
|
||||
[--pseudonymise] [--root DIR] [--out DIR]
|
||||
python tools/bugreport.py --bundle DIR --candidates FILE
|
||||
|
||||
The bundle is `<out>/bugreport-<UTC stamp>/` plus a zip beside it, in four
|
||||
layers: the environment, the stack, what `wikitool` prints (if it starts), and
|
||||
the agent's chronology. Secrets are always removed. Page titles are kept out of
|
||||
everything this script generates unless `--titles` is given. Nothing is uploaded.
|
||||
everything this script generates unless `--titles` is given. With `--pseudonymise`
|
||||
known identities are replaced by consistent, shape-preserving placeholders (stage 1);
|
||||
`--bundle`/`--candidates` applies the names a model found beyond them (stage 2). The
|
||||
mapping stays beside the bundle, never in it. Nothing is uploaded.
|
||||
|
||||
Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script
|
||||
opens no gate. See instructions/bug-report.md.
|
||||
@@ -22,11 +26,16 @@ from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import datetime
|
||||
import getpass
|
||||
import hashlib
|
||||
import hmac
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
import re
|
||||
import secrets
|
||||
import shutil
|
||||
import socket
|
||||
import subprocess
|
||||
import sys
|
||||
import unicodedata
|
||||
@@ -78,6 +87,43 @@ PRIVACY_NOTICE = (
|
||||
"yourself; neither this tool nor the agent uploads anything."
|
||||
)
|
||||
|
||||
STAGE1_NOTICE = (
|
||||
"Pseudonymisation: stage 1 applied, stage 2 not yet applied. Identities this script could read "
|
||||
"from the machine (user, host, home and repository paths, git identity, remotes) are replaced by "
|
||||
"consistent placeholders that keep length, spaces, hyphens, character classes, separators and depth. "
|
||||
"Names it cannot know - people, companies, customers, internal hosts, projects - may remain, above "
|
||||
"all in the trace, the chronology and transcripts. Secrets are removed. Read the bundle before you "
|
||||
"share it. Choose the channel yourself; neither this tool nor the agent uploads anything."
|
||||
)
|
||||
|
||||
RESIDUAL_NOTICE = (
|
||||
"Pseudonymisation: stage 1 and stage 2 applied. Stage 2 is the judgement of a model: it can miss "
|
||||
"names, companies, hosts and projects, above all in the free text of a trace or transcript over "
|
||||
"100 KB, which the model did not read in full. The forms - lengths, character classes, separators "
|
||||
"and depth - are kept on purpose, so a rare name can still be recognisable by its shape. Read the "
|
||||
"bundle before you share it. Choose the channel yourself; neither this tool nor the agent uploads "
|
||||
"anything."
|
||||
)
|
||||
|
||||
PUBLIC_ORIGIN = "gitea.nehmer.net/torben/chemenu" # same origin as chemenu/version.py; this script imports nothing from chemenu
|
||||
MIN_IDENTITY = 3
|
||||
NON_ASCII_LOWER = "äöüéèêàáâçñõøåæ"
|
||||
WORD = re.compile(r"[^\W_]+")
|
||||
|
||||
VOCABULARY = frozenset(
|
||||
w.lower() for w in (
|
||||
"Windows System32 SysWOW64 Program Files ProgramData Users Public AppData Local LocalLow "
|
||||
"Roaming Temp Microsoft WindowsApps Programs Documents Desktop Downloads OneDrive "
|
||||
"DESKTOP LAPTOP "
|
||||
"home usr local bin opt etc var tmp mnt src Library Applications "
|
||||
"Python Git Scripts venv chocolatey "
|
||||
"chemenu wikitool tools reports kb raw bugreport "
|
||||
"GmbH AG KG SE Inc Ltd LLC "
|
||||
"removed collected title depth len space "
|
||||
"json jsonl md txt py ps1 zip cfg template"
|
||||
).split()
|
||||
)
|
||||
|
||||
WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE)
|
||||
WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]')
|
||||
WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]")
|
||||
@@ -215,6 +261,328 @@ def _title_flags(title: str) -> str:
|
||||
return " ".join(flags)
|
||||
|
||||
|
||||
# ------------------------------------------------------------------------ pseudonymisation
|
||||
|
||||
|
||||
def _words(text: str):
|
||||
return [m.group(0).lower() for m in WORD.finditer(text)]
|
||||
|
||||
|
||||
class Pseudonymiser:
|
||||
"""Consistent, shape-preserving replacement of identities, word by word.
|
||||
|
||||
A placeholder hangs on the word, not on the identity: "Beispiel" is replaced the
|
||||
same way inside "OneDrive - Beispiel GmbH" (stage 1) and inside the candidate
|
||||
"Beispiel GmbH" (stage 2), so the two stages agree without coordinating.
|
||||
"""
|
||||
|
||||
def __init__(self, salt=None) -> None:
|
||||
self.salt = salt or secrets.token_hex(16)
|
||||
self.words = {}
|
||||
self.keep = set()
|
||||
self.entries = []
|
||||
self.skipped = 0
|
||||
self.stage = 1
|
||||
self._index = {}
|
||||
self._taken = set(VOCABULARY)
|
||||
self._regex = {}
|
||||
self._forms = {}
|
||||
|
||||
# -- persistence
|
||||
|
||||
def to_json(self, bundle_name: str) -> dict:
|
||||
return {
|
||||
"schema": 1, "bundle": bundle_name, "stage": self.stage, "salt": self.salt,
|
||||
"keep": sorted(self.keep), "skipped": self.skipped, "words": self.words,
|
||||
"entries": self.entries,
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, data: dict) -> "Pseudonymiser":
|
||||
pz = cls(data["salt"])
|
||||
pz.stage = data.get("stage", 1)
|
||||
pz.keep = set(data.get("keep", []))
|
||||
pz.skipped = data.get("skipped", 0)
|
||||
pz.words = dict(data.get("words", {}))
|
||||
pz.entries = list(data.get("entries", []))
|
||||
for key, pseudo in pz.words.items():
|
||||
pz._taken.add(key)
|
||||
pz._taken.add(pseudo)
|
||||
for index, entry in enumerate(pz.entries):
|
||||
pz._index[entry["original"].lower()] = index
|
||||
pz._taken.update(_words(entry["original"]))
|
||||
return pz
|
||||
|
||||
# -- placeholders
|
||||
|
||||
def _stream(self, key: str, counter: int):
|
||||
block = 0
|
||||
salt = bytes.fromhex(self.salt)
|
||||
while True:
|
||||
message = "%s\x00%d\x00%d" % (key, counter, block)
|
||||
for byte in hmac.new(salt, message.encode("utf-8"), hashlib.sha256).digest():
|
||||
yield byte
|
||||
block += 1
|
||||
|
||||
def pseudo_word(self, word: str) -> str:
|
||||
key = word.lower()
|
||||
if key in VOCABULARY or key in self.keep:
|
||||
return word
|
||||
base = self.words.get(key)
|
||||
if base is None:
|
||||
counter = 0
|
||||
while True:
|
||||
stream = self._stream(key, counter)
|
||||
chars = []
|
||||
for char in word:
|
||||
byte = next(stream)
|
||||
if char.isdigit():
|
||||
chars.append(str(byte % 10))
|
||||
elif char.isascii():
|
||||
chars.append(chr(ord("a") + byte % 26))
|
||||
else:
|
||||
chars.append(NON_ASCII_LOWER[byte % len(NON_ASCII_LOWER)])
|
||||
base = "".join(chars)
|
||||
if (base != key and base not in self._taken) or counter >= 500:
|
||||
break
|
||||
counter += 1
|
||||
self.words[key] = base
|
||||
self._taken.add(base)
|
||||
if len(base) != len(word):
|
||||
base = (base * (len(word) // max(len(base), 1) + 1))[:len(word)]
|
||||
return "".join(
|
||||
b.upper() if orig.isupper() and not b.isdigit() else b for orig, b in zip(word, base)
|
||||
)
|
||||
|
||||
def pseudo_text(self, text: str) -> str:
|
||||
return WORD.sub(lambda m: self.pseudo_word(m.group(0)), text)
|
||||
|
||||
# -- identities
|
||||
|
||||
def add_identity(self, original: str, kind: str, stage: int = 1) -> bool:
|
||||
original = original.strip()
|
||||
if not original or original.lower() in self._index:
|
||||
return False
|
||||
words = _words(original)
|
||||
if len(original) < MIN_IDENTITY or not words:
|
||||
self.skipped += 1
|
||||
return False
|
||||
if kind in ("host", "domain", "remote") and "." in original:
|
||||
label = _words(original)[-1]
|
||||
if not label.isdigit():
|
||||
self.keep.add(label)
|
||||
if all(w in VOCABULARY or w in self.keep for w in words):
|
||||
self.skipped += 1
|
||||
return False
|
||||
self._taken.update(words)
|
||||
self._index[original.lower()] = len(self.entries)
|
||||
self.entries.append({"original": original, "placeholder": None, "kind": kind,
|
||||
"stage": stage, "count": 0})
|
||||
self._regex = {}
|
||||
return True
|
||||
|
||||
def finish(self) -> None:
|
||||
for entry in self.entries:
|
||||
if entry["placeholder"] is None:
|
||||
entry["placeholder"] = self.pseudo_text(entry["original"])
|
||||
|
||||
# -- applying
|
||||
|
||||
def _build(self, stage) -> None:
|
||||
self.finish()
|
||||
forms = {PUBLIC_ORIGIN.lower(): (None, False)}
|
||||
for entry in self.entries:
|
||||
if stage is not None and entry["stage"] != stage:
|
||||
continue
|
||||
raw = entry["original"]
|
||||
forms.setdefault(raw.lower(), (entry, False))
|
||||
escaped = json.dumps(raw, ensure_ascii=True)[1:-1]
|
||||
if escaped.lower() != raw.lower():
|
||||
forms.setdefault(escaped.lower(), (entry, True))
|
||||
names = sorted(forms, key=len, reverse=True)
|
||||
self._regex[stage] = (forms, re.compile(
|
||||
r"(?<![^\W_])(?:%s)(?![^\W_])" % "|".join(re.escape(n) for n in names), re.IGNORECASE
|
||||
))
|
||||
|
||||
def apply(self, text: str, stage=None) -> str:
|
||||
if not any(stage is None or e["stage"] == stage for e in self.entries):
|
||||
return text
|
||||
if stage not in self._regex:
|
||||
self._build(stage)
|
||||
forms, regex = self._regex[stage]
|
||||
|
||||
def repl(match):
|
||||
found = forms.get(match.group(0).lower())
|
||||
if found is None or found[0] is None:
|
||||
return match.group(0)
|
||||
entry, escaped = found
|
||||
if escaped:
|
||||
try:
|
||||
decoded = json.loads('"%s"' % match.group(0))
|
||||
except ValueError:
|
||||
return match.group(0)
|
||||
entry["count"] += 1
|
||||
return json.dumps(self.pseudo_text(decoded), ensure_ascii=True)[1:-1]
|
||||
entry["count"] += 1
|
||||
return self.pseudo_text(match.group(0))
|
||||
|
||||
return regex.sub(repl, text)
|
||||
|
||||
def find(self, text: str, original: str) -> int:
|
||||
raw = re.escape(original)
|
||||
escaped = re.escape(json.dumps(original, ensure_ascii=True)[1:-1])
|
||||
pattern = re.compile(r"(?<![^\W_])(?:%s|%s)(?![^\W_])" % (raw, escaped), re.IGNORECASE)
|
||||
return len(pattern.findall(text))
|
||||
|
||||
def placeholder_words(self) -> set:
|
||||
return set(self.words.values())
|
||||
|
||||
|
||||
def _path_parts(value: str):
|
||||
return [part for part in re.split(r"[\\/]+", value) if part]
|
||||
|
||||
|
||||
def _remote_identities(url: str):
|
||||
"""(original, kind) pairs for one remote URL. The stack's public origin is exempt."""
|
||||
url = url.strip()
|
||||
if not url or PUBLIC_ORIGIN in url.lower().replace("\\", "/"):
|
||||
return []
|
||||
found = []
|
||||
scheme = re.match(r"^[A-Za-z][A-Za-z0-9+.\-]*://", url)
|
||||
scp = re.match(r"^(?:[^@/\\:\s]+@)?([^/\\:\s]{2,}):(?![/\\])(.*)$", url)
|
||||
if scheme:
|
||||
rest = url[scheme.end():]
|
||||
netloc, _, path = rest.partition("/")
|
||||
host = netloc.rsplit("@", 1)[-1].split(":")[0]
|
||||
found.append((host, "remote"))
|
||||
parts = _path_parts(path)
|
||||
elif scp:
|
||||
found.append((scp.group(1), "remote"))
|
||||
parts = _path_parts(scp.group(2))
|
||||
else:
|
||||
parts = _path_parts(url)
|
||||
found.extend((part, "remote") for part in parts)
|
||||
return found
|
||||
|
||||
|
||||
def collect_identities(root: Path):
|
||||
"""Identities readable from this machine, in the order they are registered."""
|
||||
found = []
|
||||
user = set()
|
||||
try:
|
||||
user.add(getpass.getuser())
|
||||
except Exception: # noqa: BLE001 - no account name is not a failure
|
||||
pass
|
||||
for name in ("USER", "USERNAME", "LOGNAME"):
|
||||
if os.environ.get(name):
|
||||
user.add(os.environ[name])
|
||||
found.extend((value, "user name") for value in sorted(user))
|
||||
if os.environ.get("USERDOMAIN"):
|
||||
found.append((os.environ["USERDOMAIN"], "domain"))
|
||||
if os.environ.get("COMPUTERNAME"):
|
||||
found.append((os.environ["COMPUTERNAME"], "host"))
|
||||
try:
|
||||
found.append((socket.gethostname(), "host"))
|
||||
except OSError:
|
||||
pass
|
||||
found.extend((part, "home path") for part in _path_parts(os.path.expanduser("~")))
|
||||
found.extend((part, "repo path") for part in _path_parts(str(root)))
|
||||
for key in ("user.name", "user.email"):
|
||||
result = run(["git", "config", "--get", key], root)
|
||||
value = result.stdout.strip() if result.ok else ""
|
||||
if value:
|
||||
found.append((value, "git identity"))
|
||||
if key == "user.email" and "@" in value:
|
||||
local, _, domain = value.rpartition("@")
|
||||
found.append((local, "git identity"))
|
||||
found.append((domain, "domain"))
|
||||
remotes = run(["git", "remote", "-v"], root)
|
||||
if remotes.ok:
|
||||
for line in remotes.stdout.splitlines():
|
||||
fields = line.split("\t", 1)
|
||||
if len(fields) == 2:
|
||||
found.extend(_remote_identities(re.sub(r"\s+\((?:fetch|push)\)\s*$", "", fields[1])))
|
||||
return found
|
||||
|
||||
|
||||
def text_files(bundle_dir: Path):
|
||||
return [p for p in sorted(bundle_dir.rglob("*")) if p.is_file()]
|
||||
|
||||
|
||||
def pseudonymise_bundle(bundle_dir: Path, pz: Pseudonymiser, stage=None) -> None:
|
||||
for path in text_files(bundle_dir):
|
||||
text = read_text(path)
|
||||
changed = pz.apply(text, stage)
|
||||
if changed != text:
|
||||
write_text(path, changed)
|
||||
|
||||
|
||||
EMAIL_RE = re.compile(r"[\w.+\-]+@[\w\-]+(?:\.[\w\-]+)+")
|
||||
URL_RE = re.compile(r"[A-Za-z][A-Za-z0-9+.\-]*://[^\s\"'<>)\]]+")
|
||||
PATH_RE = re.compile(r"(?:[A-Za-z]:)?(?:[\\/]+[^\\/\s\"'<>|:*?,;()\[\]{}]+){2,}")
|
||||
|
||||
|
||||
def build_review(bundle_dir: Path, pz: Pseudonymiser) -> str:
|
||||
"""What stage 1 leaves structured and unmasked, for the model that does stage 2."""
|
||||
found = {}
|
||||
|
||||
def note(kind: str, value: str, rel: str) -> None:
|
||||
words = _words(value)
|
||||
if len(value) < MIN_IDENTITY or not words:
|
||||
return
|
||||
if all(w in VOCABULARY or w in pz.keep or w in pz.placeholder_words() for w in words):
|
||||
return
|
||||
entry = found.setdefault((kind, value), {"count": 0, "files": []})
|
||||
entry["count"] += 1
|
||||
if rel not in entry["files"]:
|
||||
entry["files"].append(rel)
|
||||
|
||||
for path in text_files(bundle_dir):
|
||||
rel = path.relative_to(bundle_dir).as_posix()
|
||||
text = read_text(path)
|
||||
for match in EMAIL_RE.finditer(text):
|
||||
note("mail address", match.group(0), rel)
|
||||
for match in URL_RE.finditer(text):
|
||||
rest = match.group(0).split("://", 1)[1]
|
||||
netloc, _, tail = rest.partition("/")
|
||||
note("host", netloc.rsplit("@", 1)[-1], rel)
|
||||
for part in _path_parts(tail):
|
||||
note("url segment", part, rel)
|
||||
for match in PATH_RE.finditer(text):
|
||||
for part in _path_parts(match.group(0)):
|
||||
note("path component", part, rel)
|
||||
lines = [
|
||||
"# Review list for stage 2 - local only, never part of the bundle. It holds what stage 1",
|
||||
"# left in the bundle, which can still be a real name. One line: kind, count, value, files.",
|
||||
]
|
||||
for (kind, value), info in sorted(found.items(), key=lambda kv: (-kv[1]["count"], kv[0])):
|
||||
lines.append("%s\t%d\t%s\t%s" % (kind, info["count"], value, ", ".join(info["files"][:5])))
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def pseudonym_table(pz: Pseudonymiser) -> str:
|
||||
lines = ["| Placeholder | Kind | Stage | Occurrences |", "|---|---|---|---|"]
|
||||
for entry in pz.entries:
|
||||
placeholder = entry["placeholder"].replace("|", "\\|")
|
||||
lines.append("| `%s` | %s | %d | %d |" % (
|
||||
placeholder, entry["kind"], entry["stage"], entry["count"]))
|
||||
lines += ["", "%d identities were left unchanged (under %d characters, or system vocabulary); "
|
||||
"their values are not listed." % (pz.skipped, MIN_IDENTITY)]
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def replace_section(manifest: str, heading: str, body: str) -> str:
|
||||
sections = re.split(r"(?m)^(?=## )", manifest)
|
||||
block = "## %s\n\n%s\n" % (heading, body)
|
||||
for index, section in enumerate(sections):
|
||||
if section.startswith("## %s\n" % heading):
|
||||
sections[index] = block
|
||||
break
|
||||
else:
|
||||
sections.append(block)
|
||||
return "\n\n".join(part.strip("\n") for part in sections if part.strip()) + "\n"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------- helpers
|
||||
|
||||
|
||||
@@ -751,11 +1119,14 @@ def unique_stamp(out: Path) -> str:
|
||||
def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str:
|
||||
lines = ["# Bug report bundle", "",
|
||||
"Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"),
|
||||
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", PRIVACY_NOTICE, "",
|
||||
"Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "",
|
||||
STAGE1_NOTICE if args.pseudonymise else PRIVACY_NOTICE, "",
|
||||
"## Options", "",
|
||||
"- chronology: %s" % ("supplied" if args.chronology else "not supplied"),
|
||||
"- transcripts: %d" % len(args.transcript),
|
||||
"- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"),
|
||||
] + (["- pseudonymisation (`--pseudonymise`): stage 1 applied, stage 2 not yet applied"]
|
||||
if args.pseudonymise else []) + [
|
||||
"- session trace: %s" % (
|
||||
"excluded (`--no-trace`)" if args.no_trace else
|
||||
("included" if trace and trace.get("found") else "not found")),
|
||||
@@ -827,13 +1198,27 @@ def parse_args(argv):
|
||||
group.add_argument("--no-trace", action="store_true", help="leave the session trace out")
|
||||
parser.add_argument("--titles", action="store_true",
|
||||
help="keep page titles and paths (writes tree-paths.txt)")
|
||||
parser.add_argument("--pseudonymise", action="store_true",
|
||||
help="replace known identities by consistent, shape-preserving "
|
||||
"placeholders (stage 1); the mapping stays beside the bundle")
|
||||
parser.add_argument("--root", help="checkout root (default: parent of tools/)")
|
||||
parser.add_argument("--out", help="output directory (default: <root>/reports)")
|
||||
return parser.parse_args(argv)
|
||||
parser.add_argument("--bundle", help="stage 2: an existing pseudonymised bundle directory")
|
||||
parser.add_argument("--candidates", help="stage 2: file with one further name per line "
|
||||
"('#' starts a comment)")
|
||||
args = parser.parse_args(argv)
|
||||
if bool(args.bundle) != bool(args.candidates):
|
||||
parser.error("--bundle and --candidates belong together")
|
||||
if args.bundle and (args.chronology or args.transcript or args.session or args.no_trace
|
||||
or args.titles or args.pseudonymise or args.root or args.out):
|
||||
parser.error("--bundle/--candidates excludes every collection option")
|
||||
return args
|
||||
|
||||
|
||||
def main(argv=None) -> int:
|
||||
args = parse_args(argv)
|
||||
if args.bundle:
|
||||
return run_stage2(Path(args.bundle), Path(args.candidates))
|
||||
root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent
|
||||
out = Path(args.out).resolve() if args.out else root / "reports"
|
||||
|
||||
@@ -921,13 +1306,114 @@ def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int:
|
||||
)
|
||||
final_pass(bundle_dir, red, shield, {"tree-paths.txt"})
|
||||
|
||||
pz = None
|
||||
if args.pseudonymise:
|
||||
pz = Pseudonymiser()
|
||||
for original, kind in collect_identities(root):
|
||||
pz.add_identity(original, kind)
|
||||
pz.finish()
|
||||
pseudonymise_bundle(bundle_dir, pz)
|
||||
manifest_path = bundle_dir / "MANIFEST.md"
|
||||
write_text(manifest_path, replace_section(
|
||||
read_text(manifest_path), "Pseudonyms", pseudonym_table(pz)))
|
||||
write_text(_sibling(bundle_dir, ".pseudonyms.json"), dump_json(pz.to_json(bundle_dir.name)))
|
||||
write_text(_sibling(bundle_dir, ".review.txt"), build_review(bundle_dir, pz))
|
||||
|
||||
archive = out / ("bugreport-%s.zip" % stamp)
|
||||
make_zip(bundle_dir, archive)
|
||||
|
||||
print("Bundle: %s" % bundle_dir)
|
||||
print("Archive: %s" % archive)
|
||||
if pz is not None:
|
||||
print("Mapping: %s" % _sibling(bundle_dir, ".pseudonyms.json"))
|
||||
print("Review: %s" % _sibling(bundle_dir, ".review.txt"))
|
||||
print("Both hold originals and stay on this machine: never share them.")
|
||||
print("Stage 2: python3 tools/bugreport.py --bundle %s --candidates <file>" % bundle_dir)
|
||||
print()
|
||||
print(PRIVACY_NOTICE)
|
||||
print(STAGE1_NOTICE if pz is not None else PRIVACY_NOTICE)
|
||||
return 0
|
||||
|
||||
|
||||
def _sibling(bundle_dir: Path, suffix: str) -> Path:
|
||||
return bundle_dir.with_name(bundle_dir.name + suffix)
|
||||
|
||||
|
||||
def parse_candidates(text: str):
|
||||
names = []
|
||||
for line in text.splitlines():
|
||||
line = re.sub(r"(^|\s)#.*$", "", line).strip()
|
||||
if line and line not in names:
|
||||
names.append(line)
|
||||
return names
|
||||
|
||||
|
||||
def run_stage2(bundle_dir: Path, candidates: Path) -> int:
|
||||
bundle_dir = bundle_dir.resolve()
|
||||
mapping = _sibling(bundle_dir, ".pseudonyms.json")
|
||||
problem = None
|
||||
if not bundle_dir.is_dir():
|
||||
problem = "bundle directory not found: %s" % bundle_dir
|
||||
elif not candidates.is_file():
|
||||
problem = "candidates file not found: %s" % candidates
|
||||
elif not mapping.is_file():
|
||||
problem = ("no mapping beside the bundle (%s): stage 2 needs the salt stage 1 used, so "
|
||||
"collect the report again with --pseudonymise" % mapping.name)
|
||||
if problem:
|
||||
sys.stderr.write("bugreport: %s\n" % problem)
|
||||
return 1
|
||||
try:
|
||||
pz = Pseudonymiser.from_json(json.loads(read_text(mapping)))
|
||||
except (ValueError, KeyError) as exc:
|
||||
sys.stderr.write("bugreport: the mapping cannot be read: %s\n" % exc)
|
||||
return 1
|
||||
|
||||
files = text_files(bundle_dir)
|
||||
corpus = "\n".join(read_text(path) for path in files)
|
||||
applied, skipped = [], []
|
||||
placeholders = pz.placeholder_words()
|
||||
for name in parse_candidates(read_text(candidates)):
|
||||
words = _words(name)
|
||||
if len(name) < MIN_IDENTITY or not words:
|
||||
skipped.append((name, "under %d characters" % MIN_IDENTITY))
|
||||
elif all(w in VOCABULARY or w in pz.keep for w in words):
|
||||
skipped.append((name, "system vocabulary"))
|
||||
elif all(w in placeholders for w in words):
|
||||
skipped.append((name, "already a placeholder"))
|
||||
elif pz.find(corpus, name) == 0:
|
||||
skipped.append((name, "not found in the bundle"))
|
||||
else:
|
||||
applied.append(name)
|
||||
try:
|
||||
for name in applied:
|
||||
pz.add_identity(name, "stage-2 candidate", stage=2)
|
||||
pz.stage = 2
|
||||
pseudonymise_bundle(bundle_dir, pz, stage=2)
|
||||
manifest_path = bundle_dir / "MANIFEST.md"
|
||||
manifest = read_text(manifest_path)
|
||||
manifest = replace_section(manifest, "Privacy", RESIDUAL_NOTICE)
|
||||
manifest = replace_section(manifest, "Pseudonyms", pseudonym_table(pz))
|
||||
manifest = re.sub(r"(?m)^- pseudonymisation \(`--pseudonymise`\): .*$",
|
||||
"- pseudonymisation (`--pseudonymise`): stage 1 and stage 2 applied",
|
||||
manifest)
|
||||
write_text(manifest_path, manifest)
|
||||
write_text(mapping, dump_json(pz.to_json(bundle_dir.name)))
|
||||
archive = bundle_dir.with_name(bundle_dir.name + ".zip")
|
||||
temporary = archive.with_name(archive.name + ".tmp")
|
||||
make_zip(bundle_dir, temporary)
|
||||
os.replace(str(temporary), str(archive))
|
||||
except OSError as exc:
|
||||
sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc)
|
||||
return 1
|
||||
|
||||
print("Bundle: %s" % bundle_dir)
|
||||
print("Archive: %s" % archive)
|
||||
print("Mapping: %s (holds originals; delete it after the last stage 2 run)" % mapping)
|
||||
for name in applied:
|
||||
print("applied: %s" % name)
|
||||
for name, reason in skipped:
|
||||
print("not applied: %s - %s" % (name, reason))
|
||||
print()
|
||||
print(RESIDUAL_NOTICE)
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
Reference in new issue
Block a user