diff --git a/.gitea/workflows/ci.yml b/.gitea/workflows/ci.yml index e75f860..4e0b57f 100644 --- a/.gitea/workflows/ci.yml +++ b/.gitea/workflows/ci.yml @@ -294,3 +294,23 @@ jobs: test -f "$bundle/environment.json" grep -q '`wikitool` started' "$bundle/MANIFEST.md" ls reports/bugreport-*.zip + # The same collector with --pseudonymise: the mapping lies beside the + # bundle, not in it or its zip, and this container's hostname is gone. + rm -rf reports/bugreport-* + python3 tools/bugreport.py --no-trace --pseudonymise + bundle=$(ls -d reports/bugreport-*/ | head -n 1) + bundle=${bundle%/} + test -f "$bundle.pseudonyms.json" + test -f "$bundle.review.txt" + grep -q 'stage 1 applied, stage 2 not yet applied' "$bundle/MANIFEST.md" + python3 - "$bundle" <<'PY' + import socket, sys, zipfile + from pathlib import Path + bundle = Path(sys.argv[1]) + names = zipfile.ZipFile(str(bundle) + ".zip").namelist() + assert not [n for n in names if "pseudonyms" in n or "review" in n], names + host = socket.gethostname() + for path in bundle.rglob("*"): + if path.is_file(): + assert host not in path.read_text(encoding="utf-8", errors="replace"), path + PY diff --git a/CHANGES.md b/CHANGES.md index 962e32a..62005e7 100644 --- a/CHANGES.md +++ b/CHANGES.md @@ -59,7 +59,7 @@ concern - readable here, never shipped as something to parse. --- -## 8.0.0-beta.7 - 2026-09-30 - reports/CONTRACT.md: only the collector's two counting calls take the bugreport session id +## 8.0.0-beta.8 - 2026-09-30 - Bug-report collector can pseudonymise identities, in two stages **Author:** Torben Nehmer @@ -81,6 +81,7 @@ concern - readable here, never shipped as something to parse. - Super Productivity API path: unwrap the {ok, data} envelope, exclude the inbox project, ready-aware health (#162) - Live tracker suite: WIKITOOL_TASKS_CONFIG override, real-tracker tests for Super Productivity and CalDAV, nightly workflow and test image - Bug-report collector: tools/bugreport.py and instructions/bug-report.md +- Bug-report collector can pseudonymise identities, in two stages **Low impact** - version bump no longer points at version release in its output @@ -114,6 +115,37 @@ concern - readable here, never shipped as something to parse. - reports/CONTRACT.md: only the collector's two counting calls take the bugreport session id +### Bug-report collector can pseudonymise identities, in two stages (Gitea #158) + +A bundle carries machine, user and path names, and an installation failure usually turns on the +*shape* of those names, not on the names. `tools/bugreport.py --pseudonymise` therefore replaces +each identity by a placeholder of the same shape and leaves the structure alone. It is opt-in, +costs one more step, and the instruction recommends it for every channel except a direct handover +to the maintainer over a secure channel. + +- **Stage 1 is mechanical.** The script reads user, `USERDOMAIN`/`COMPUTERNAME`, hostname, home and + repository path, `git config user.name`/`user.email` and the remote URLs, and replaces each word + of them in every text file, in raw and JSON-escaped form. A placeholder is an HMAC-SHA256 of the + lowercased word under a per-bundle random salt, so it keeps length, digit/ASCII/non-ASCII class + and per-occurrence case, is injective through a counter, and differs between two bundles. Words of + the stack's own vocabulary, top-level domains and the public origin stay readable. Matching is + whole-identity, longest first, on word boundaries. +- **Stage 2 is a model's judgement, applied by the script.** The agent reads the review list plus + `CHRONOLOGY.md` and `MANIFEST.md` in full, and trace and transcripts in full only under 100 KB + per file, and names further people, companies, customers, internal hosts and projects in a + candidate file. `--bundle DIR --candidates FILE` applies them with the same machinery and packs + the zip again; the model replaces nothing itself. It refuses with exit 1 and an unchanged bundle + when the mapping is gone, and can be repeated. +- **Three local files** - the mapping with its salt, the review list and the candidate file - sit + beside the bundle directory, never inside it and never in the zip. All three hold originals. +- **`MANIFEST.md` names three privacy states** (none, stage 1, stage 1 and 2) and lists the + placeholders, never an original. The closing output and the instruction name the residual + uncertainty: stage 2 can miss a name in free text it did not read in full. +- `instructions/bug-report.md`, `reports/CONTRACT.md`, `tools/README.md` and `INSTALL.md` describe + both stages and the three files; CI runs the collector with `--pseudonymise` from the exported + distribution and checks that the mapping is beside, not in, the bundle and that the host name is + gone. + ### reports/CONTRACT.md: only the collector's two counting calls take the bugreport session id The contract said all of the collector's `wikitool` calls run under `bugreport-`. Only diff --git a/INSTALL.md b/INSTALL.md index 72379de..2f95a76 100644 --- a/INSTALL.md +++ b/INSTALL.md @@ -480,8 +480,13 @@ tools/wikitool instructions verify samt Zip. Geheimnisse werden entfernt, und aus allem, was das Skript selbst erzeugt, bleiben Seitentitel draußen (`--titles` nimmt sie mit). Der Sitzungs-Trace ist standardmäßig dabei (`--no-trace` lässt ihn weg) und kann wie Chronologie und Transkripte Seiteninhalt und Titel - enthalten; das Bündel enthält außerdem Maschinen-, Benutzer- und Pfadnamen. Es wird nirgends hochgeladen - - den Kanal wählst du selbst. + enthalten; das Bündel enthält außerdem Maschinen-, Benutzer- und Pfadnamen. Mit `--pseudonymise` + ersetzt das Skript Benutzer-, Host-, Pfad-, Git- und Remote-Namen durch Platzhalter, die Länge, + Leerzeichen, Bindestriche und Pfadtiefe erhalten; der Agent kann danach weitere Namen + (Personen, Firmen, Kunden, Projekte) mit `--bundle … --candidates …` nachtragen. Das ist das + Urteil eines Modells und lässt einen Rest übrig - lies das Bündel vor dem Teilen. Die + Zuordnung, die Prüfliste und die Kandidatendatei enthalten Originale und liegen neben, nie im + Bündel. Es wird nirgends hochgeladen - den Kanal wählst du selbst. - **Ich will am Tool-Stack selbst weiterarbeiten (nicht nur Wiki-Inhalt betreiben)** - eine neue Instanz hat dafür keinen Weg: `dist export` lässt `instructions/dev/` (Stack-Entwicklung, inkl. der vendorten `commonplace/`-Wissensbasis) bewusst und dauerhaft weg, ohne diff --git a/VERSION b/VERSION index de86ae9..bc4d1e9 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -8.0.0-beta.7 +8.0.0-beta.8 diff --git a/instructions/bug-report.md b/instructions/bug-report.md index a26fd2c..9b4da3f 100644 --- a/instructions/bug-report.md +++ b/instructions/bug-report.md @@ -53,6 +53,26 @@ out** unless `--titles` is given: paths under `kb/` and `raw/` are replaced by t flags. A title that stands as bare prose is not found - which is why the trace, the chronology and the transcripts are marked instead. +**Pseudonymisation is optional** (`--pseudonymise`) and runs in two stages. The bundle then keeps the +*shape* of every name - length per word, spaces, hyphens, character classes, separators, depth of a +path - because that is what an installation failure turns on, and replaces the name itself. + +- **Stage 1 is mechanical.** The script reads what the machine knows about its user - user name, + host, home and repository path, git identity, remote URLs - and replaces each identity in every + text file by a placeholder. The same word always gets the same placeholder, in every file and in + JSON-escaped form too. The stack's own public origin and system folder names stay readable. +- **Stage 2 is a model's judgement, applied mechanically.** Names the script cannot know - people, + companies, customers, internal hosts, projects - are named by you as candidates in a file; the + script applies them with the same machinery. You replace nothing in the bundle yourself. +- **Three local files** sit beside the bundle directory, never inside it and never in the zip: + `bugreport-.pseudonyms.json` (the mapping), `bugreport-.review.txt` (what stage 1 + left behind, for you to read) and the candidate file you write. All three contain originals. +- **A residual uncertainty remains and is always named:** stage 2 can miss a name, above all in the + free text of a large trace or transcript that you did not read in full. + +The model of the harness reads the bundle for stage 2 - the same place that already sees this +session. The bundle does not leave that place because of it. + ## When to run - Setup or an upgrade failed and the user wants to report it. @@ -66,6 +86,10 @@ transcripts are marked instead. remotes, and it is not sent anywhere. Ask whether the session trace, page titles and transcripts may go in. The defaults are: trace in, titles out, no transcripts. + Ask which channel the bundle will take, and recommend pseudonymisation for every channel except a + direct handover to the maintainer over a secure channel: a tracker issue, an email or a chat is + not one. The default is off; say so, and that it costs one more step. + 2. **Write the chronology** to a file under `reports/` (which is gitignored) from the [template](#chronology-template) below. Facts only - no diagnosis. If the session's own history is too long to reconstruct, say what is missing instead of filling the gap. @@ -78,17 +102,35 @@ transcripts are marked instead. Add `--no-trace` if the user declined the trace, `--titles` if titles may stay, `--transcript ` (repeatable) for a transcript the user pointed at, `--session ` to take a - trace other than the caller's. `python3` may be `python` or `py -3` on Windows. If no Python starts + trace other than the caller's, `--pseudonymise` if the user chose it (stage 1). `python3` may be `python` or `py -3` on Windows. If no Python starts at all, that is the report: give the user the exact error text, verbatim. -4. **Do not imitate the collector's session id.** It runs its counting `wikitool` calls under +4. **Stage 2, only if the bundle was pseudonymised.** Read, in the bundle: `CHRONOLOGY.md` and + `MANIFEST.md` completely; the review list `reports/bugreport-.review.txt` completely; the + trace and each transcript completely only if the file is under 100 KB, otherwise only what the + review list points to. Name what is still left of a person, company, customer, internal host or + domain, or project - one per line in `reports/candidates.txt`; `#` starts a comment. Then run: + + ```bash + python3 tools/bugreport.py --bundle reports/bugreport- --candidates reports/candidates.txt + ``` + + The script applies the candidates with the machinery of stage 1, reports which it did not apply + (too short, system vocabulary, not found) and packs the zip again. **Never replace anything in + the bundle yourself.** The run can be repeated with further candidates. It refuses, with exit 1 and + an unchanged bundle, when the mapping beside the bundle is gone. + +5. **Do not imitate the collector's session id.** It runs its counting `wikitool` calls under `WIKITOOL_SESSION_ID=bugreport-` itself; do not export that variable, or any `bugreport-*` one, in the session. The exception it gets in [gates.md](gates.md) § "Taking a new session id" belongs to the script alone. -5. **Report the result.** Quote the bundle path, the archive path and the privacy notice the script +6. **Report the result.** Quote the bundle path, the archive path and the privacy notice the script prints, name the gaps in `MANIFEST.md`, and tell the user to read the bundle before sharing it. - Then stop: the channel - a tracker issue, an email, a chat - is the user's choice, and the agent + After stage 2, name the residual uncertainty in words of your own: stage 2 is a model's judgement + and can have missed names, above all in the free text of a trace or transcript over 100 KB. + Say that the mapping, the review list and the candidate file hold originals and stay on this + machine, and that the mapping may be deleted after the last stage 2 run. Then stop: the channel - a tracker issue, an email, a chat - is the user's choice, and the agent never uploads the bundle. ## Chronology template @@ -121,6 +163,11 @@ Facts only: no page content, no guessed cause, no advice. ## Decision points +- **The mapping is gone before stage 2?** Stage 2 refuses: it needs the salt stage 1 used, or it would + replace a word differently from the path it already replaced. Collect the report again with + `--pseudonymise`. +- **The user wants a name added after stage 2?** Run stage 2 again with a candidate file holding only + that line. The same run also works when the human spots a name while reading the bundle. - **The user declines the trace?** Run with `--no-trace`. The manifest records the exclusion. - **The failure can be reproduced, and there is no trace?** A distributed instance records none by default. Offer to repeat the failing step in one new shell with `WIKI_TRACE=1` set for that diff --git a/reports/CONTRACT.md b/reports/CONTRACT.md index e230129..bfc3f6b 100644 --- a/reports/CONTRACT.md +++ b/reports/CONTRACT.md @@ -50,6 +50,14 @@ session trace, the chronology and transcripts, which may hold page content and t removed by the collector; the rest is the reader's to check before a bundle leaves the machine. Nothing uploads it: the channel is the user's choice. +With `--pseudonymise` the collector replaces known identities by consistent, shape-preserving +placeholders, and `--bundle`/`--candidates` applies further names a model found. Three files then +sit **beside** the bundle directory, never inside it and never in its zip, and all three contain +originals: `bugreport-.pseudonyms.json` (the mapping and its salt), +`bugreport-.review.txt` (what stage 1 left behind) and the candidate file the agent writes. +`MANIFEST.md` lists the placeholders, never an original. What stage 2 finds is a model's judgement, +so the manifest and the collector's closing output name a residual uncertainty. + The collector's two counting `wikitool` calls (`instructions verify`, `docs verify`) run under the session id `bugreport-`; its budget-exempt ones inherit the caller's. A `reports/telemetry/bugreport-*` directory is therefore that run's trace and belongs to no session @@ -62,7 +70,9 @@ retire with `wikitool rm`, because no report is ever a wiki page - `lint-report` contract-only type-spec with no `base_dir:` and cannot be instantiated under `kb/`. **Bug-report bundles: none.** Delete them freely once they have been read or sent; nothing refers -to one afterwards. +to one afterwards. The mapping beside a pseudonymised bundle is needed until the last stage 2 run, +because stage 2 refuses without it; after that it may be deleted, and it should be, since it holds +the originals. **Traces: two enforced caps, applied by the writer itself, never by a separate cleanup pass.** A byte cap per session trace (default 5 MiB, `WIKI_TRACE_MAX_SESSION_BYTES`) and a retention diff --git a/tools/README.md b/tools/README.md index ee1b678..3ea0e92 100644 --- a/tools/README.md +++ b/tools/README.md @@ -64,7 +64,11 @@ tools/ from it: it is the bug-report collector ([../instructions/bug-report.md](../instructions/bug-report.md)), and the case it exists for is a checkout where `wikitool` does not start. It therefore uses the standard library only, keeps to Python 3.8 syntax so that an old interpreter can still run it, and -exits 0 or 1 - never 42, since it opens no gate. `dist export` ships it with the rest of `tools/`; +exits 0 or 1 - never 42, since it opens no gate. It has a second mode: `--pseudonymise` replaces the +identities it can read from the machine (stage 1), and `--bundle`/`--candidates` applies names a +model found in the result (stage 2), both word by word so that length, separators and depth +survive. The mapping, the review list and the candidate file stay beside the bundle directory. +`dist export` ships it with the rest of `tools/`; its tests (`tests/test_bugreport.py`) run it on a bare interpreter (`-I -S`) to keep that promise. **Two consumers, one core.** The CLI is not the only caller any more. The cores diff --git a/tools/bugreport.py b/tools/bugreport.py index abfde20..8dd2aa4 100644 --- a/tools/bugreport.py +++ b/tools/bugreport.py @@ -8,12 +8,16 @@ Runs on the base Python with the standard library only, and imports nothing from python tools/bugreport.py [--chronology FILE] [--transcript FILE]... [--session ID | --no-trace] [--titles] - [--root DIR] [--out DIR] + [--pseudonymise] [--root DIR] [--out DIR] + python tools/bugreport.py --bundle DIR --candidates FILE The bundle is `/bugreport-/` plus a zip beside it, in four layers: the environment, the stack, what `wikitool` prints (if it starts), and the agent's chronology. Secrets are always removed. Page titles are kept out of -everything this script generates unless `--titles` is given. Nothing is uploaded. +everything this script generates unless `--titles` is given. With `--pseudonymise` +known identities are replaced by consistent, shape-preserving placeholders (stage 1); +`--bundle`/`--candidates` applies the names a model found beyond them (stage 2). The +mapping stays beside the bundle, never in it. Nothing is uploaded. Exit 0 = bundle written, 1 = it could not be written. No exit 42: this script opens no gate. See instructions/bug-report.md. @@ -22,11 +26,16 @@ from __future__ import annotations import argparse import datetime +import getpass +import hashlib +import hmac import json import os import platform import re +import secrets import shutil +import socket import subprocess import sys import unicodedata @@ -78,6 +87,43 @@ PRIVACY_NOTICE = ( "yourself; neither this tool nor the agent uploads anything." ) +STAGE1_NOTICE = ( + "Pseudonymisation: stage 1 applied, stage 2 not yet applied. Identities this script could read " + "from the machine (user, host, home and repository paths, git identity, remotes) are replaced by " + "consistent placeholders that keep length, spaces, hyphens, character classes, separators and depth. " + "Names it cannot know - people, companies, customers, internal hosts, projects - may remain, above " + "all in the trace, the chronology and transcripts. Secrets are removed. Read the bundle before you " + "share it. Choose the channel yourself; neither this tool nor the agent uploads anything." +) + +RESIDUAL_NOTICE = ( + "Pseudonymisation: stage 1 and stage 2 applied. Stage 2 is the judgement of a model: it can miss " + "names, companies, hosts and projects, above all in the free text of a trace or transcript over " + "100 KB, which the model did not read in full. The forms - lengths, character classes, separators " + "and depth - are kept on purpose, so a rare name can still be recognisable by its shape. Read the " + "bundle before you share it. Choose the channel yourself; neither this tool nor the agent uploads " + "anything." +) + +PUBLIC_ORIGIN = "gitea.nehmer.net/torben/chemenu" # same origin as chemenu/version.py; this script imports nothing from chemenu +MIN_IDENTITY = 3 +NON_ASCII_LOWER = "äöüéèêàáâçñõøåæ" +WORD = re.compile(r"[^\W_]+") + +VOCABULARY = frozenset( + w.lower() for w in ( + "Windows System32 SysWOW64 Program Files ProgramData Users Public AppData Local LocalLow " + "Roaming Temp Microsoft WindowsApps Programs Documents Desktop Downloads OneDrive " + "DESKTOP LAPTOP " + "home usr local bin opt etc var tmp mnt src Library Applications " + "Python Git Scripts venv chocolatey " + "chemenu wikitool tools reports kb raw bugreport " + "GmbH AG KG SE Inc Ltd LLC " + "removed collected title depth len space " + "json jsonl md txt py ps1 zip cfg template" + ).split() +) + WINDOWS_RESERVED = re.compile(r"^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$", re.IGNORECASE) WINDOWS_FORBIDDEN = re.compile(r'[<>:"|?*\x00-\x1f]') WIKILINK = re.compile(r"\[\[([^\]\n]+)\]\]") @@ -215,6 +261,328 @@ def _title_flags(title: str) -> str: return " ".join(flags) +# ------------------------------------------------------------------------ pseudonymisation + + +def _words(text: str): + return [m.group(0).lower() for m in WORD.finditer(text)] + + +class Pseudonymiser: + """Consistent, shape-preserving replacement of identities, word by word. + + A placeholder hangs on the word, not on the identity: "Beispiel" is replaced the + same way inside "OneDrive - Beispiel GmbH" (stage 1) and inside the candidate + "Beispiel GmbH" (stage 2), so the two stages agree without coordinating. + """ + + def __init__(self, salt=None) -> None: + self.salt = salt or secrets.token_hex(16) + self.words = {} + self.keep = set() + self.entries = [] + self.skipped = 0 + self.stage = 1 + self._index = {} + self._taken = set(VOCABULARY) + self._regex = {} + self._forms = {} + + # -- persistence + + def to_json(self, bundle_name: str) -> dict: + return { + "schema": 1, "bundle": bundle_name, "stage": self.stage, "salt": self.salt, + "keep": sorted(self.keep), "skipped": self.skipped, "words": self.words, + "entries": self.entries, + } + + @classmethod + def from_json(cls, data: dict) -> "Pseudonymiser": + pz = cls(data["salt"]) + pz.stage = data.get("stage", 1) + pz.keep = set(data.get("keep", [])) + pz.skipped = data.get("skipped", 0) + pz.words = dict(data.get("words", {})) + pz.entries = list(data.get("entries", [])) + for key, pseudo in pz.words.items(): + pz._taken.add(key) + pz._taken.add(pseudo) + for index, entry in enumerate(pz.entries): + pz._index[entry["original"].lower()] = index + pz._taken.update(_words(entry["original"])) + return pz + + # -- placeholders + + def _stream(self, key: str, counter: int): + block = 0 + salt = bytes.fromhex(self.salt) + while True: + message = "%s\x00%d\x00%d" % (key, counter, block) + for byte in hmac.new(salt, message.encode("utf-8"), hashlib.sha256).digest(): + yield byte + block += 1 + + def pseudo_word(self, word: str) -> str: + key = word.lower() + if key in VOCABULARY or key in self.keep: + return word + base = self.words.get(key) + if base is None: + counter = 0 + while True: + stream = self._stream(key, counter) + chars = [] + for char in word: + byte = next(stream) + if char.isdigit(): + chars.append(str(byte % 10)) + elif char.isascii(): + chars.append(chr(ord("a") + byte % 26)) + else: + chars.append(NON_ASCII_LOWER[byte % len(NON_ASCII_LOWER)]) + base = "".join(chars) + if (base != key and base not in self._taken) or counter >= 500: + break + counter += 1 + self.words[key] = base + self._taken.add(base) + if len(base) != len(word): + base = (base * (len(word) // max(len(base), 1) + 1))[:len(word)] + return "".join( + b.upper() if orig.isupper() and not b.isdigit() else b for orig, b in zip(word, base) + ) + + def pseudo_text(self, text: str) -> str: + return WORD.sub(lambda m: self.pseudo_word(m.group(0)), text) + + # -- identities + + def add_identity(self, original: str, kind: str, stage: int = 1) -> bool: + original = original.strip() + if not original or original.lower() in self._index: + return False + words = _words(original) + if len(original) < MIN_IDENTITY or not words: + self.skipped += 1 + return False + if kind in ("host", "domain", "remote") and "." in original: + label = _words(original)[-1] + if not label.isdigit(): + self.keep.add(label) + if all(w in VOCABULARY or w in self.keep for w in words): + self.skipped += 1 + return False + self._taken.update(words) + self._index[original.lower()] = len(self.entries) + self.entries.append({"original": original, "placeholder": None, "kind": kind, + "stage": stage, "count": 0}) + self._regex = {} + return True + + def finish(self) -> None: + for entry in self.entries: + if entry["placeholder"] is None: + entry["placeholder"] = self.pseudo_text(entry["original"]) + + # -- applying + + def _build(self, stage) -> None: + self.finish() + forms = {PUBLIC_ORIGIN.lower(): (None, False)} + for entry in self.entries: + if stage is not None and entry["stage"] != stage: + continue + raw = entry["original"] + forms.setdefault(raw.lower(), (entry, False)) + escaped = json.dumps(raw, ensure_ascii=True)[1:-1] + if escaped.lower() != raw.lower(): + forms.setdefault(escaped.lower(), (entry, True)) + names = sorted(forms, key=len, reverse=True) + self._regex[stage] = (forms, re.compile( + r"(? str: + if not any(stage is None or e["stage"] == stage for e in self.entries): + return text + if stage not in self._regex: + self._build(stage) + forms, regex = self._regex[stage] + + def repl(match): + found = forms.get(match.group(0).lower()) + if found is None or found[0] is None: + return match.group(0) + entry, escaped = found + if escaped: + try: + decoded = json.loads('"%s"' % match.group(0)) + except ValueError: + return match.group(0) + entry["count"] += 1 + return json.dumps(self.pseudo_text(decoded), ensure_ascii=True)[1:-1] + entry["count"] += 1 + return self.pseudo_text(match.group(0)) + + return regex.sub(repl, text) + + def find(self, text: str, original: str) -> int: + raw = re.escape(original) + escaped = re.escape(json.dumps(original, ensure_ascii=True)[1:-1]) + pattern = re.compile(r"(? set: + return set(self.words.values()) + + +def _path_parts(value: str): + return [part for part in re.split(r"[\\/]+", value) if part] + + +def _remote_identities(url: str): + """(original, kind) pairs for one remote URL. The stack's public origin is exempt.""" + url = url.strip() + if not url or PUBLIC_ORIGIN in url.lower().replace("\\", "/"): + return [] + found = [] + scheme = re.match(r"^[A-Za-z][A-Za-z0-9+.\-]*://", url) + scp = re.match(r"^(?:[^@/\\:\s]+@)?([^/\\:\s]{2,}):(?![/\\])(.*)$", url) + if scheme: + rest = url[scheme.end():] + netloc, _, path = rest.partition("/") + host = netloc.rsplit("@", 1)[-1].split(":")[0] + found.append((host, "remote")) + parts = _path_parts(path) + elif scp: + found.append((scp.group(1), "remote")) + parts = _path_parts(scp.group(2)) + else: + parts = _path_parts(url) + found.extend((part, "remote") for part in parts) + return found + + +def collect_identities(root: Path): + """Identities readable from this machine, in the order they are registered.""" + found = [] + user = set() + try: + user.add(getpass.getuser()) + except Exception: # noqa: BLE001 - no account name is not a failure + pass + for name in ("USER", "USERNAME", "LOGNAME"): + if os.environ.get(name): + user.add(os.environ[name]) + found.extend((value, "user name") for value in sorted(user)) + if os.environ.get("USERDOMAIN"): + found.append((os.environ["USERDOMAIN"], "domain")) + if os.environ.get("COMPUTERNAME"): + found.append((os.environ["COMPUTERNAME"], "host")) + try: + found.append((socket.gethostname(), "host")) + except OSError: + pass + found.extend((part, "home path") for part in _path_parts(os.path.expanduser("~"))) + found.extend((part, "repo path") for part in _path_parts(str(root))) + for key in ("user.name", "user.email"): + result = run(["git", "config", "--get", key], root) + value = result.stdout.strip() if result.ok else "" + if value: + found.append((value, "git identity")) + if key == "user.email" and "@" in value: + local, _, domain = value.rpartition("@") + found.append((local, "git identity")) + found.append((domain, "domain")) + remotes = run(["git", "remote", "-v"], root) + if remotes.ok: + for line in remotes.stdout.splitlines(): + fields = line.split("\t", 1) + if len(fields) == 2: + found.extend(_remote_identities(re.sub(r"\s+\((?:fetch|push)\)\s*$", "", fields[1]))) + return found + + +def text_files(bundle_dir: Path): + return [p for p in sorted(bundle_dir.rglob("*")) if p.is_file()] + + +def pseudonymise_bundle(bundle_dir: Path, pz: Pseudonymiser, stage=None) -> None: + for path in text_files(bundle_dir): + text = read_text(path) + changed = pz.apply(text, stage) + if changed != text: + write_text(path, changed) + + +EMAIL_RE = re.compile(r"[\w.+\-]+@[\w\-]+(?:\.[\w\-]+)+") +URL_RE = re.compile(r"[A-Za-z][A-Za-z0-9+.\-]*://[^\s\"'<>)\]]+") +PATH_RE = re.compile(r"(?:[A-Za-z]:)?(?:[\\/]+[^\\/\s\"'<>|:*?,;()\[\]{}]+){2,}") + + +def build_review(bundle_dir: Path, pz: Pseudonymiser) -> str: + """What stage 1 leaves structured and unmasked, for the model that does stage 2.""" + found = {} + + def note(kind: str, value: str, rel: str) -> None: + words = _words(value) + if len(value) < MIN_IDENTITY or not words: + return + if all(w in VOCABULARY or w in pz.keep or w in pz.placeholder_words() for w in words): + return + entry = found.setdefault((kind, value), {"count": 0, "files": []}) + entry["count"] += 1 + if rel not in entry["files"]: + entry["files"].append(rel) + + for path in text_files(bundle_dir): + rel = path.relative_to(bundle_dir).as_posix() + text = read_text(path) + for match in EMAIL_RE.finditer(text): + note("mail address", match.group(0), rel) + for match in URL_RE.finditer(text): + rest = match.group(0).split("://", 1)[1] + netloc, _, tail = rest.partition("/") + note("host", netloc.rsplit("@", 1)[-1], rel) + for part in _path_parts(tail): + note("url segment", part, rel) + for match in PATH_RE.finditer(text): + for part in _path_parts(match.group(0)): + note("path component", part, rel) + lines = [ + "# Review list for stage 2 - local only, never part of the bundle. It holds what stage 1", + "# left in the bundle, which can still be a real name. One line: kind, count, value, files.", + ] + for (kind, value), info in sorted(found.items(), key=lambda kv: (-kv[1]["count"], kv[0])): + lines.append("%s\t%d\t%s\t%s" % (kind, info["count"], value, ", ".join(info["files"][:5]))) + return "\n".join(lines) + "\n" + + +def pseudonym_table(pz: Pseudonymiser) -> str: + lines = ["| Placeholder | Kind | Stage | Occurrences |", "|---|---|---|---|"] + for entry in pz.entries: + placeholder = entry["placeholder"].replace("|", "\\|") + lines.append("| `%s` | %s | %d | %d |" % ( + placeholder, entry["kind"], entry["stage"], entry["count"])) + lines += ["", "%d identities were left unchanged (under %d characters, or system vocabulary); " + "their values are not listed." % (pz.skipped, MIN_IDENTITY)] + return "\n".join(lines) + + +def replace_section(manifest: str, heading: str, body: str) -> str: + sections = re.split(r"(?m)^(?=## )", manifest) + block = "## %s\n\n%s\n" % (heading, body) + for index, section in enumerate(sections): + if section.startswith("## %s\n" % heading): + sections[index] = block + break + else: + sections.append(block) + return "\n\n".join(part.strip("\n") for part in sections if part.strip()) + "\n" + + # ---------------------------------------------------------------------------- helpers @@ -751,11 +1119,14 @@ def unique_stamp(out: Path) -> str: def build_manifest(args, root, stamp, bundle, wikitool, trace, gaps) -> str: lines = ["# Bug report bundle", "", "Collected: %s UTC " % datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%d %H:%M:%S"), - "Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", PRIVACY_NOTICE, "", + "Bundle: `bugreport-%s`" % stamp, "", "## Privacy", "", + STAGE1_NOTICE if args.pseudonymise else PRIVACY_NOTICE, "", "## Options", "", "- chronology: %s" % ("supplied" if args.chronology else "not supplied"), "- transcripts: %d" % len(args.transcript), "- page titles (`--titles`): %s" % ("included" if args.titles else "kept out"), + ] + (["- pseudonymisation (`--pseudonymise`): stage 1 applied, stage 2 not yet applied"] + if args.pseudonymise else []) + [ "- session trace: %s" % ( "excluded (`--no-trace`)" if args.no_trace else ("included" if trace and trace.get("found") else "not found")), @@ -827,13 +1198,27 @@ def parse_args(argv): group.add_argument("--no-trace", action="store_true", help="leave the session trace out") parser.add_argument("--titles", action="store_true", help="keep page titles and paths (writes tree-paths.txt)") + parser.add_argument("--pseudonymise", action="store_true", + help="replace known identities by consistent, shape-preserving " + "placeholders (stage 1); the mapping stays beside the bundle") parser.add_argument("--root", help="checkout root (default: parent of tools/)") parser.add_argument("--out", help="output directory (default: /reports)") - return parser.parse_args(argv) + parser.add_argument("--bundle", help="stage 2: an existing pseudonymised bundle directory") + parser.add_argument("--candidates", help="stage 2: file with one further name per line " + "('#' starts a comment)") + args = parser.parse_args(argv) + if bool(args.bundle) != bool(args.candidates): + parser.error("--bundle and --candidates belong together") + if args.bundle and (args.chronology or args.transcript or args.session or args.no_trace + or args.titles or args.pseudonymise or args.root or args.out): + parser.error("--bundle/--candidates excludes every collection option") + return args def main(argv=None) -> int: args = parse_args(argv) + if args.bundle: + return run_stage2(Path(args.bundle), Path(args.candidates)) root = Path(args.root).resolve() if args.root else Path(__file__).resolve().parent.parent out = Path(args.out).resolve() if args.out else root / "reports" @@ -921,13 +1306,114 @@ def _collect(args, root: Path, out: Path, stamp: str, bundle_dir: Path) -> int: ) final_pass(bundle_dir, red, shield, {"tree-paths.txt"}) + pz = None + if args.pseudonymise: + pz = Pseudonymiser() + for original, kind in collect_identities(root): + pz.add_identity(original, kind) + pz.finish() + pseudonymise_bundle(bundle_dir, pz) + manifest_path = bundle_dir / "MANIFEST.md" + write_text(manifest_path, replace_section( + read_text(manifest_path), "Pseudonyms", pseudonym_table(pz))) + write_text(_sibling(bundle_dir, ".pseudonyms.json"), dump_json(pz.to_json(bundle_dir.name))) + write_text(_sibling(bundle_dir, ".review.txt"), build_review(bundle_dir, pz)) + archive = out / ("bugreport-%s.zip" % stamp) make_zip(bundle_dir, archive) print("Bundle: %s" % bundle_dir) print("Archive: %s" % archive) + if pz is not None: + print("Mapping: %s" % _sibling(bundle_dir, ".pseudonyms.json")) + print("Review: %s" % _sibling(bundle_dir, ".review.txt")) + print("Both hold originals and stay on this machine: never share them.") + print("Stage 2: python3 tools/bugreport.py --bundle %s --candidates " % bundle_dir) print() - print(PRIVACY_NOTICE) + print(STAGE1_NOTICE if pz is not None else PRIVACY_NOTICE) + return 0 + + +def _sibling(bundle_dir: Path, suffix: str) -> Path: + return bundle_dir.with_name(bundle_dir.name + suffix) + + +def parse_candidates(text: str): + names = [] + for line in text.splitlines(): + line = re.sub(r"(^|\s)#.*$", "", line).strip() + if line and line not in names: + names.append(line) + return names + + +def run_stage2(bundle_dir: Path, candidates: Path) -> int: + bundle_dir = bundle_dir.resolve() + mapping = _sibling(bundle_dir, ".pseudonyms.json") + problem = None + if not bundle_dir.is_dir(): + problem = "bundle directory not found: %s" % bundle_dir + elif not candidates.is_file(): + problem = "candidates file not found: %s" % candidates + elif not mapping.is_file(): + problem = ("no mapping beside the bundle (%s): stage 2 needs the salt stage 1 used, so " + "collect the report again with --pseudonymise" % mapping.name) + if problem: + sys.stderr.write("bugreport: %s\n" % problem) + return 1 + try: + pz = Pseudonymiser.from_json(json.loads(read_text(mapping))) + except (ValueError, KeyError) as exc: + sys.stderr.write("bugreport: the mapping cannot be read: %s\n" % exc) + return 1 + + files = text_files(bundle_dir) + corpus = "\n".join(read_text(path) for path in files) + applied, skipped = [], [] + placeholders = pz.placeholder_words() + for name in parse_candidates(read_text(candidates)): + words = _words(name) + if len(name) < MIN_IDENTITY or not words: + skipped.append((name, "under %d characters" % MIN_IDENTITY)) + elif all(w in VOCABULARY or w in pz.keep for w in words): + skipped.append((name, "system vocabulary")) + elif all(w in placeholders for w in words): + skipped.append((name, "already a placeholder")) + elif pz.find(corpus, name) == 0: + skipped.append((name, "not found in the bundle")) + else: + applied.append(name) + try: + for name in applied: + pz.add_identity(name, "stage-2 candidate", stage=2) + pz.stage = 2 + pseudonymise_bundle(bundle_dir, pz, stage=2) + manifest_path = bundle_dir / "MANIFEST.md" + manifest = read_text(manifest_path) + manifest = replace_section(manifest, "Privacy", RESIDUAL_NOTICE) + manifest = replace_section(manifest, "Pseudonyms", pseudonym_table(pz)) + manifest = re.sub(r"(?m)^- pseudonymisation \(`--pseudonymise`\): .*$", + "- pseudonymisation (`--pseudonymise`): stage 1 and stage 2 applied", + manifest) + write_text(manifest_path, manifest) + write_text(mapping, dump_json(pz.to_json(bundle_dir.name))) + archive = bundle_dir.with_name(bundle_dir.name + ".zip") + temporary = archive.with_name(archive.name + ".tmp") + make_zip(bundle_dir, temporary) + os.replace(str(temporary), str(archive)) + except OSError as exc: + sys.stderr.write("bugreport: cannot write the bundle: %s\n" % exc) + return 1 + + print("Bundle: %s" % bundle_dir) + print("Archive: %s" % archive) + print("Mapping: %s (holds originals; delete it after the last stage 2 run)" % mapping) + for name in applied: + print("applied: %s" % name) + for name, reason in skipped: + print("not applied: %s - %s" % (name, reason)) + print() + print(RESIDUAL_NOTICE) return 0 diff --git a/tools/chemenu/tests/test_bugreport.py b/tools/chemenu/tests/test_bugreport.py index 84b9456..4d74520 100644 --- a/tools/chemenu/tests/test_bugreport.py +++ b/tools/chemenu/tests/test_bugreport.py @@ -325,3 +325,234 @@ def test_the_collector_ships_with_a_distribution(): from chemenu.commands import dist_cmd assert "tools/bugreport.py" in dist_cmd.build_plan() + + +# ------------------------------------------------------------------ pseudonymisation + +REMOTE = "C:\\Users\\Max.Muster\\OneDrive - Beispiel GmbH\\x.git" +REPO_ROOT = Path(SCRIPT).resolve().parent.parent + + +def machine(checkout: Path, monkeypatch, user: str = "Max") -> None: + for name in ("USER", "USERNAME", "LOGNAME"): + monkeypatch.setenv(name, user) + monkeypatch.setenv("COMPUTERNAME", "DESKTOP-Zx91Q") + monkeypatch.setenv("USERDOMAIN", "AB") + monkeypatch.setattr(bugreport.socket, "gethostname", lambda: "Rechner-Delta") + git(checkout, "config", "user.name", "Erika Mustermann") + git(checkout, "config", "user.email", "erika@beispiel-firma.example") + remotes = subprocess.run(["git", "remote"], cwd=checkout, capture_output=True, text=True).stdout + if "corp" not in remotes.split(): + git(checkout, "remote", "add", "corp", REMOTE) + + +def pseudonymised(checkout, tmp_path, monkeypatch, *extra, user="Max", name="out"): + machine(checkout, monkeypatch, user) + out = tmp_path / name + assert bugreport.main( + ["--root", str(checkout), "--out", str(out), "--no-trace", "--pseudonymise", *extra] + ) == 0 + return next(p for p in out.iterdir() if p.is_dir()) + + +def zip_text(bundle: Path) -> str: + archive = bundle.with_name(bundle.name + ".zip") + with zipfile.ZipFile(archive) as zf: + return "\n".join(zf.read(n).decode("utf-8", errors="replace") for n in zf.namelist()) + + +def mapping_of(bundle: Path) -> Path: + return bundle.with_name(bundle.name + ".pseudonyms.json") + + +def snapshot(bundle: Path) -> dict: + return {p.relative_to(bundle).as_posix(): p.read_bytes() for p in bundle.rglob("*") if p.is_file()} + + +def test_a_remote_keeps_its_shape_and_loses_its_names(checkout, tmp_path, monkeypatch): + bundle = pseudonymised(checkout, tmp_path, monkeypatch) + assert "Max.Muster" not in all_text(bundle) and "Max.Muster" not in zip_text(bundle) + assert "Beispiel" not in all_text(bundle) + remotes = json.loads((bundle / "environment.json").read_text())["git"]["remotes"] + line = next(r for r in remotes if r.startswith("corp")) + original = f"corp\t{REMOTE} (fetch)".split("\\") + shaped = line.split("\\") + assert len(shaped) == len(original) + assert [len(part) for part in shaped] == [len(part) for part in original] + assert shaped[1] == "Users" + assert shaped[2][3] == "." and shaped[2] != "Max.Muster" + assert shaped[3].startswith("OneDrive - ") and shaped[3].endswith(" GmbH") + assert shaped[4].endswith(".git (fetch)") + + +def test_one_identity_gets_one_placeholder_in_every_file_and_form(checkout, tmp_path, monkeypatch): + name = "Müller-Lüdenscheidt" + monkeypatch.setenv("PATH", "/home/%s/bin%s%s" % (name, os.pathsep, os.environ["PATH"])) + trace_dir = checkout / "reports" / "telemetry" / "callers-session" + trace_dir.mkdir(parents=True) + (trace_dir / "trace.jsonl").write_text(json.dumps({"p": "/home/" + name}) + "\n") + monkeypatch.setenv("WIKITOOL_SESSION_ID", "callers-session") + chron = tmp_path / "chron.md" + chron.write_text("user %s and %s\n" % (name, name.upper())) + machine(checkout, monkeypatch, user=name) + out = tmp_path / "out" + assert bugreport.main(["--root", str(checkout), "--out", str(out), "--pseudonymise", + "--chronology", str(chron)]) == 0 + bundle = next(p for p in out.iterdir() if p.is_dir()) + env_entry = json.loads((bundle / "environment.json").read_text())["path_entries"][0]["entry"] + from_env = env_entry.split("/")[2] + from_trace = json.loads((bundle / "trace.jsonl").read_text())["p"].split("/")[2] + lines = (bundle / "CHRONOLOGY.md").read_text().split() + assert from_env == from_trace == lines[1] + assert lines[3] == from_env.upper() + assert from_env != name and len(from_env) == len(name) + for original, shaped in zip(name, from_env): + assert (original == "-") == (shaped == "-") + assert original.isascii() == shaped.isascii() + assert original.isupper() == shaped.isupper() + assert name not in all_text(bundle) and "M\\u00fcller" not in all_text(bundle) + + +def test_placeholders_keep_length_classes_and_separators_and_are_injective(): + pz = bugreport.Pseudonymiser(salt="00" * 16) + assert pz.add_identity("Müller-Lüdenscheidt 42", "user name") + pz.finish() + placeholder = pz.entries[0]["placeholder"] + original = pz.entries[0]["original"] + assert len(placeholder) == len(original) + for a, b in zip(original, placeholder): + assert a.isalpha() == b.isalpha() and a.isdigit() == b.isdigit() + assert a.isascii() == b.isascii() and a.isupper() == b.isupper() + if not a.isalnum(): + assert a == b + assert pz.pseudo_text("Users GmbH") == "Users GmbH" + words = ["name%03d" % i for i in range(300)] + shaped = {pz.pseudo_word(w) for w in words} + assert len(shaped) == len(words) + assert not shaped & set(words) + + +def test_an_identity_matches_whole_words_only(checkout, tmp_path, monkeypatch): + chron = tmp_path / "chron.md" + chron.write_text("Maximum and Max_1 and Max here, Maxine too\n") + bundle = pseudonymised(checkout, tmp_path, monkeypatch, "--chronology", str(chron)) + text = (bundle / "CHRONOLOGY.md").read_text() + assert "Maximum" in text and "Maxine" in text + assert text.count("Max") == 2 # "Maximum" and "Maxine" contain none at a word end; Max_1 and Max are replaced + + +def test_stage_two_applies_candidates_with_the_same_word_and_is_idempotent( + checkout, tmp_path, monkeypatch, capsys +): + chron = tmp_path / "chron.md" + chron.write_text("Firma Beispiel GmbH hat Zugriff. Kunde Acme Corp. Acme Corp zahlt.\n") + bundle = pseudonymised(checkout, tmp_path, monkeypatch, "--chronology", str(chron)) + assert "Beispiel GmbH" in (bundle / "CHRONOLOGY.md").read_text() + remotes = json.loads((bundle / "environment.json").read_text())["git"]["remotes"] + stage_one_word = next(r for r in remotes if r.startswith("corp")).split("\\")[3].split()[2] + candidates = tmp_path / "candidates.txt" + candidates.write_text("# from the model\nBeispiel GmbH\nAcme Corp # customer\nab\nUsers\nnowhere at all\n") + capsys.readouterr() + assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 0 + printed = capsys.readouterr().out + for needle in ("not applied: ab", "not applied: Users", "not applied: nowhere at all", + "100 KB"): + assert needle in printed + chronology = (bundle / "CHRONOLOGY.md").read_text() + assert "Acme" not in chronology and "Beispiel" not in chronology + assert chronology.startswith("Firma %s GmbH hat" % stage_one_word) + assert "Acme" not in zip_text(bundle) and "Beispiel" not in zip_text(bundle) + first = snapshot(bundle) + mapping_before = mapping_of(bundle).read_bytes() + assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 0 + assert snapshot(bundle) == first + assert mapping_of(bundle).read_bytes() == mapping_before + + +def test_the_mapping_review_and_candidates_stay_beside_the_bundle_and_stage_two_needs_the_mapping( + checkout, tmp_path, monkeypatch, capsys +): + bundle = pseudonymised(checkout, tmp_path, monkeypatch) + review = bundle.with_name(bundle.name + ".review.txt") + assert mapping_of(bundle).is_file() and review.is_file() + assert "Max.Muster" in mapping_of(bundle).read_text() + archive = bundle.with_name(bundle.name + ".zip") + with zipfile.ZipFile(archive) as zf: + assert not [n for n in zf.namelist() if "pseudonyms" in n or "review" in n] + assert not [p for p in bundle.rglob("*") if "pseudonyms" in p.name or "review" in p.name] + + candidates = tmp_path / "candidates.txt" + candidates.write_text("Erika\n") + mapping_of(bundle).unlink() + before, archive_before = snapshot(bundle), archive.read_bytes() + assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 1 + assert "collect the report again" in capsys.readouterr().err + assert snapshot(bundle) == before and archive.read_bytes() == archive_before + + +def test_the_manifest_has_three_privacy_versions_and_never_an_original( + checkout, tmp_path, monkeypatch +): + plain = collect(checkout, tmp_path / "plain", "--no-trace") + assert "not pseudonymised" in (plain / "MANIFEST.md").read_text() + assert "## Pseudonyms" not in (plain / "MANIFEST.md").read_text() + + bundle = pseudonymised(checkout, tmp_path, monkeypatch) + manifest = (bundle / "MANIFEST.md").read_text() + assert "stage 1 applied, stage 2 not yet applied" in manifest + assert "| Placeholder | Kind | Stage | Occurrences |" in manifest + for original in ("Max.Muster", "Beispiel", "Rechner-Delta", "Mustermann", "beispiel-firma"): + assert original not in manifest + + candidates = tmp_path / "candidates.txt" + candidates.write_text("Erika\n") + assert bugreport.main(["--bundle", str(bundle), "--candidates", str(candidates)]) == 0 + manifest = (bundle / "MANIFEST.md").read_text() + assert "stage 1 and stage 2 applied" in manifest + assert "judgement of a model" in manifest and "100 KB" in manifest + assert "Erika" not in manifest + instruction = (REPO_ROOT / "instructions" / "bug-report.md").read_text() + assert "100 KB" in instruction and "residual" in instruction.lower() + + +def test_two_bundles_of_one_machine_use_different_placeholders(checkout, tmp_path, monkeypatch): + monkeypatch.setattr(bugreport, "stamp_now", lambda: "20260101T000000Z") + first = pseudonymised(checkout, tmp_path, monkeypatch, name="a") + second = pseudonymised(checkout, tmp_path, monkeypatch, name="b") + + def shaped(bundle): + remotes = json.loads((bundle / "environment.json").read_text())["git"]["remotes"] + return next(r for r in remotes if r.startswith("corp")).split("\\")[2] + + assert shaped(first) != shaped(second) + + +def test_the_public_origin_short_and_vocabulary_identities_stay_readable( + checkout, tmp_path, monkeypatch +): + git(checkout, "remote", "add", "pub", "https://gitea.nehmer.net/torben/chemenu.git") + chron = tmp_path / "chron.md" + chron.write_text("torben was here; AB and DESKTOP stay\n") + bundle = pseudonymised(checkout, tmp_path, monkeypatch, "--chronology", str(chron), user="torben") + text = (bundle / "environment.json").read_text() + assert "gitea.nehmer.net/torben/chemenu.git" in text + chronology = (bundle / "CHRONOLOGY.md").read_text() + assert chronology.startswith("torben was") is False + assert "AB and DESKTOP stay" in chronology + manifest = (bundle / "MANIFEST.md").read_text() + assert int(manifest.split(" identities were left unchanged")[0].split()[-1]) >= 2 + + +def test_bundle_and_candidates_exclude_the_collection_options(checkout, tmp_path): + for extra in (["--no-trace"], ["--titles"], ["--pseudonymise"], ["--root", str(checkout)]): + with pytest.raises(SystemExit): + bugreport.main(["--bundle", "b", "--candidates", "c", *extra]) + with pytest.raises(SystemExit): + bugreport.main(["--bundle", "b"]) + with pytest.raises(SystemExit): + bugreport.main(["--candidates", "c"]) + + +def test_without_the_flag_nothing_but_the_bundle_and_its_archive_is_written(checkout, tmp_path): + collect(checkout, tmp_path, "--no-trace") + assert sorted(p.suffix for p in (tmp_path / "out").iterdir()) == ["", ".zip"]