Files
chemenu/tools/chemenu/web_capture.py
T
torbenandClaude Opus 5.5 fba263af68
CI / verify (push) Successful in 5m21s
CI / pwsh (push) Successful in 2m6s
Release / release (push) Successful in 34s
feat: raw fetch - a sanctioned intake for a URL into incoming/, HTML as received plus derived text (#120)
Files changed:
- CHANGES.md
- README.md
- VERSION
- instructions/wiki-ingest/SKILL.md
- raw/CONTRACT.md
- tools/CONTRACT.md
- tools/README.md
- tools/chemenu/cli_contract.py
- tools/chemenu/commands/raw_cmd.py
- tools/chemenu/tests/test_cli.py
- tools/chemenu/tests/test_portability.py
- tools/chemenu/tests/test_raw_fetch.py
- tools/chemenu/web_capture.py

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
2026-10-02 22:26:06 +02:00

650 lines
24 KiB
Python

"""Capturing a web page for `raw/`: fetch the bytes, decide their character
set, and derive a Markdown-like reading text from them - with no CLI attached.
`wikitool raw fetch` is the terminal adapter over this module; it decides where
the files go and what the caller is told. Everything here is deterministic on
purpose: the same bytes always produce the same derived text, so two sessions
capturing the same article produce the same `raw/` bundle rather than two
hand-built variants of it. That is also why there is no readability heuristic
(text density, line numbers): a heuristic tuned once drifts the next time it is
tuned, and the received HTML is kept beside the derivation anyway, so what the
derivation drops is never lost.
Standard library only. A third-party HTML library would be a new entry in
`tools/requirements.txt`, missing from an instance's venv until someone runs
`pip install` after an upgrade - which would make this a breaking change for a
gain the kept HTML already covers.
"""
from __future__ import annotations
import codecs
import datetime
import email.message
import json
import mimetypes
import re
import time
import unicodedata
import urllib.error
import urllib.parse
import urllib.request
from dataclasses import dataclass
from html.parser import HTMLParser
from typing import Optional, Union
from chemenu.errors import BackendError, ValidationError
TIMEOUT_SECONDS = 30.0
MAX_BYTES = 25 * 1024 * 1024
SHORT_TEXT_CHARS = 200
STEM_MAX_CHARS = 60
META_SCAN_BYTES = 4096
ALLOWED_SCHEMES = ("http", "https")
HTML_TYPES = ("text/html", "application/xhtml+xml")
# Content types stored as received, under a fixed extension, without a
# derivation: they are already text a session can read and cite directly.
PLAIN_TEXT_EXTENSIONS = {"text/plain": ".txt", "text/markdown": ".md"}
# --- fetching ---------------------------------------------------------------
@dataclass(frozen=True)
class Response:
url: str
final_url: str
status: int
content_type: str
body: bytes
@property
def media_type(self) -> str:
"""The bare media type, lowercased - `text/html` out of
`text/html; charset=utf-8`; empty when the server sent none."""
return media_type(self.content_type)
def media_type(content_type: str) -> str:
if not content_type.strip():
return ""
message = email.message.Message()
message["content-type"] = content_type
return message.get_content_type().lower()
def check_url(url: str) -> None:
"""Only `http`/`https`. A `file:` URL would be a way into `incoming/` that
bypasses a human dropping the file there; `ftp:` and the rest are not
pages."""
parts = urllib.parse.urlsplit(url)
if parts.scheme.lower() not in ALLOWED_SCHEMES or not parts.netloc:
raise ValidationError(
f"{url} is not an http(s) URL - `raw fetch` fetches web pages only. A local file "
"goes into incoming/ by hand."
)
class _HttpOnlyRedirects(urllib.request.HTTPRedirectHandler):
"""urllib follows a redirect to `ftp:` on its own; the scheme rule above
has to hold for every hop, not only for the URL the caller typed."""
def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D102 - urllib's hook
check_url(urllib.parse.urljoin(req.full_url, newurl))
return super().redirect_request(req, fp, code, msg, headers, newurl)
def fetch(
url: str,
user_agent: str,
timeout: float = TIMEOUT_SECONDS,
max_bytes: int = MAX_BYTES,
) -> Response:
"""GET `url` and return the body exactly as received.
No cookies (the opener carries no cookie processor) and no content
decoding - urllib asks for `identity`, so the bytes are the page, not a
compressed transfer of it. `timeout` bounds the whole transfer, not only a
single read, so a server that trickles bytes cannot hold the call open.
Raises `ValidationError` for a URL outside `http`/`https` (also on a
redirect) and `BackendError` for everything the network does: an HTTP
error status, an unreachable host, a timeout, a body over `max_bytes`.
"""
check_url(url)
opener = urllib.request.build_opener(_HttpOnlyRedirects)
request = urllib.request.Request(url, headers={"User-Agent": user_agent, "Accept": "*/*"})
deadline = time.monotonic() + timeout
try:
with opener.open(request, timeout=timeout) as response:
declared = response.headers.get("Content-Length")
if declared and declared.strip().isdigit() and int(declared) > max_bytes:
raise BackendError(_too_large(url, max_bytes))
chunks: list[bytes] = []
size = 0
while True:
if time.monotonic() > deadline:
raise BackendError(f"{url} did not finish within {timeout:g} s.")
chunk = response.read(64 * 1024)
if not chunk:
break
size += len(chunk)
if size > max_bytes:
raise BackendError(_too_large(url, max_bytes))
chunks.append(chunk)
return Response(
url=url,
final_url=response.geturl(),
status=response.status,
content_type=response.headers.get("Content-Type", "") or "",
body=b"".join(chunks),
)
except urllib.error.HTTPError as exc:
raise BackendError(f"{url} answered HTTP {exc.code} {exc.reason}.") from exc
except urllib.error.URLError as exc:
if isinstance(exc.reason, TimeoutError):
raise BackendError(f"{url} did not answer within {timeout:g} s.") from exc
raise BackendError(f"Could not reach {url}: {exc.reason}") from exc
except TimeoutError as exc:
raise BackendError(f"{url} did not answer within {timeout:g} s.") from exc
except OSError as exc:
raise BackendError(f"Could not fetch {url}: {exc}") from exc
def _too_large(url: str, max_bytes: int) -> str:
return f"{url} is larger than {max_bytes // (1024 * 1024)} MiB - nothing was written."
def extension_for(content_type: str) -> str:
"""The file extension a non-HTML response is stored under. Read from
Python's built-in table only - `mimetypes.MimeTypes()` ignores the
machine's `/etc/mime.types`, so the answer does not depend on the host."""
kind = media_type(content_type)
if kind in PLAIN_TEXT_EXTENSIONS:
return PLAIN_TEXT_EXTENSIONS[kind]
if not kind:
return ".bin"
return mimetypes.MimeTypes().guess_extension(kind) or ".bin"
# --- character set ----------------------------------------------------------
_BOMS = (
(codecs.BOM_UTF8, "utf-8"),
(codecs.BOM_UTF32_LE, "utf-32-le"),
(codecs.BOM_UTF32_BE, "utf-32-be"),
(codecs.BOM_UTF16_LE, "utf-16-le"),
(codecs.BOM_UTF16_BE, "utf-16-be"),
)
# Covers both `<meta charset="x">` and
# `<meta http-equiv="Content-Type" content="text/html; charset=x">`.
_META_CHARSET = re.compile(rb"""<meta\b[^>]*?charset\s*=\s*["']?\s*([A-Za-z0-9_.:\-]+)""", re.IGNORECASE)
@dataclass(frozen=True)
class Decoded:
text: str
charset: str
source: str # bom | header | meta | default
replaced: int # how many U+FFFD the decoding introduced
def _known(label: Optional[str]) -> Optional[str]:
if not label:
return None
label = label.strip().lower()
try:
codecs.lookup(label)
except LookupError:
return None
return label
def decode(body: bytes, content_type: Optional[str]) -> Decoded:
"""Decode an HTML body, in a fixed order: byte-order mark, then the
`charset` of the HTTP `Content-Type` (`None` when there was no HTTP
response, as for a page saved from a browser), then a `<meta>` declaration
in the first 4 KiB, then UTF-8. A label Python does not know is skipped
like an absent one. Undecodable bytes are replaced, never dropped, and
counted, so the caller can say so."""
charset, source, start = None, "", 0
for bom, name in _BOMS:
if body.startswith(bom):
charset, source, start = name, "bom", len(bom)
break
if charset is None and content_type:
message = email.message.Message()
message["content-type"] = content_type
charset = _known(message.get_content_charset())
source = "header" if charset else ""
if charset is None:
match = _META_CHARSET.search(body[:META_SCAN_BYTES])
charset = _known(match.group(1).decode("ascii", "replace")) if match else None
source = "meta" if charset else ""
if charset is None:
charset, source = "utf-8", "default"
payload = body[start:]
try:
return Decoded(payload.decode(charset), charset, source, 0)
except UnicodeDecodeError:
text = payload.decode(charset, errors="replace")
before = payload.decode(charset, errors="ignore").count("�")
return Decoded(text, charset, source, text.count("�") - before)
# --- HTML -> text -----------------------------------------------------------
_VOID = frozenset(
"area base br col embed hr img input keygen link meta param source track wbr".split()
)
_DROPPED = frozenset(
"script style noscript nav header footer aside form template svg head title".split()
)
_BLOCK = frozenset(
"""address article blockquote body center dd details dialog div dl dt fieldset
figcaption figure h1 h2 h3 h4 h5 h6 hgroup hr html li main ol p pre section summary
table tbody thead tfoot tr td th caption ul""".split()
)
# An open `<p>` ends where one of these starts - the parser's share of the
# implied end tags real-world HTML relies on.
_CLOSES_P = frozenset(
"""address article blockquote div dl fieldset figure h1 h2 h3 h4 h5 h6 hr main ol p pre
section table ul""".split()
)
_HEADINGS = {f"h{n}": n for n in range(1, 7)}
_WS = re.compile(r"[ \t\n\r\f\v]+")
_BR = "\x00"
class _Element:
__slots__ = ("tag", "attrs", "children", "parent")
def __init__(self, tag: str, attrs: dict[str, str], parent: Optional["_Element"]):
self.tag = tag
self.attrs = attrs
self.children: list[Union["_Element", str]] = []
self.parent = parent
def iter(self):
yield self
for child in self.children:
if isinstance(child, _Element):
yield from child.iter()
def text(self) -> str:
return "".join(c if isinstance(c, str) else c.text() for c in self.children)
class _TreeBuilder(HTMLParser):
"""A forgiving tree: an end tag closes the nearest open element of that
name and is ignored when none is open, void elements never take children,
and `<p>`/`<li>`/`<dt>`/`<dd>`/`<tr>`/`<td>`/`<th>`/`<option>` end where
the next sibling of their kind begins."""
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.root = _Element("#document", {}, None)
self.stack = [self.root]
def _close_to(self, tag: str, stop_at: tuple[str, ...] = ()) -> None:
for index in range(len(self.stack) - 1, 0, -1):
name = self.stack[index].tag
if name == tag:
del self.stack[index:]
return
if name in stop_at:
return
def handle_starttag(self, tag, attrs):
if tag in _CLOSES_P:
self._close_to("p", stop_at=("div", "li", "td", "th", "blockquote", "section", "article", "main", "body"))
if tag == "li":
self._close_to("li", stop_at=("ul", "ol"))
elif tag in ("dt", "dd"):
self._close_to("dt", stop_at=("dl",))
self._close_to("dd", stop_at=("dl",))
elif tag == "tr":
self._close_to("tr", stop_at=("table",))
elif tag in ("td", "th"):
self._close_to("td", stop_at=("tr", "table"))
self._close_to("th", stop_at=("tr", "table"))
elif tag == "option":
self._close_to("option", stop_at=("select",))
element = _Element(tag, {k: (v or "") for k, v in attrs}, self.stack[-1])
self.stack[-1].children.append(element)
if tag not in _VOID:
self.stack.append(element)
def handle_startendtag(self, tag, attrs):
element = _Element(tag, {k: (v or "") for k, v in attrs}, self.stack[-1])
self.stack[-1].children.append(element)
def handle_endtag(self, tag):
if tag not in _VOID:
self._close_to(tag)
def handle_data(self, data):
self.stack[-1].children.append(data)
def parse(text: str) -> _Element:
builder = _TreeBuilder()
builder.feed(text)
builder.close()
return builder.root
def title_of(document: _Element) -> str:
for element in document.iter():
if element.tag == "title":
return _WS.sub(" ", element.text()).strip()
return ""
def content_root(document: _Element) -> _Element:
"""`<main>`, else the one `<article>` if there is exactly one, else
`<body>`, else the whole document."""
elements = list(document.iter())
for element in elements:
if element.tag == "main":
return element
articles = [e for e in elements if e.tag == "article"]
if len(articles) == 1:
return articles[0]
for element in elements:
if element.tag == "body":
return element
return document
@dataclass
class _Block:
kind: str # "text" or "list"
text: str
class _Renderer:
def __init__(self, base_url: str) -> None:
self.base_url = base_url
def url(self, href: str) -> str:
return urllib.parse.urljoin(self.base_url, href.strip())
# Blocks -------------------------------------------------------------
def blocks(self, element: _Element) -> list[_Block]:
"""The blocks of a container: runs of inline content become one
paragraph each, block children contribute their own blocks."""
out: list[_Block] = []
inline: list[str] = []
def flush() -> None:
text = _finish_inline("".join(inline))
if text:
out.append(_Block("text", text))
inline.clear()
for child in element.children:
if isinstance(child, str):
inline.append(child)
elif child.tag in _DROPPED:
continue
elif child.tag in _BLOCK:
flush()
out.extend(self.block(child))
else:
inline.append(self.inline(child))
flush()
return out
def block(self, element: _Element) -> list[_Block]:
tag = element.tag
if tag in _HEADINGS:
text = _finish_inline(self.inline_children(element))
return [_Block("text", "#" * _HEADINGS[tag] + " " + text)] if text else []
if tag == "p":
text = _finish_inline(self.inline_children(element))
return [_Block("text", text)] if text else []
if tag == "pre":
return self.pre(element)
if tag in ("ul", "ol"):
return self.list(element)
if tag == "blockquote":
inner = _join(self.blocks(element))
if not inner:
return []
return [_Block("text", "\n".join(f"> {line}" if line else ">" for line in inner.split("\n")))]
if tag == "hr":
return [_Block("text", "---")]
if tag == "table":
return self.table(element)
return self.blocks(element)
def pre(self, element: _Element) -> list[_Block]:
text = element.text().strip("\n")
if not text.strip():
return []
longest = max((len(run) for run in re.findall(r"`+", text)), default=0)
fence = "`" * max(3, longest + 1)
return [_Block("text", f"{fence}\n{text}\n{fence}")]
def list(self, element: _Element) -> list[_Block]:
ordered = element.tag == "ol"
start = element.attrs.get("start", "").strip()
number = int(start) if ordered and start.lstrip("-").isdigit() else 1
lines: list[str] = []
for child in element.children:
if isinstance(child, str) or child.tag in _DROPPED:
continue
if child.tag == "li":
content = self.blocks(child)
elif child.tag in ("ul", "ol"):
# A list nested directly in a list, without its own <li>.
content = self.block(child)
else:
continue
if not content:
continue
marker = f"{number}. " if ordered else "- "
number += 1
body = ""
for index, block in enumerate(content):
if index:
body += "\n" if block.kind == "list" else "\n\n"
body += block.text
indent = " " * len(marker)
lines.append("\n".join(
(marker if i == 0 else (indent if line else "")) + line
for i, line in enumerate(body.split("\n"))
))
return [_Block("list", "\n".join(lines))] if lines else []
def table(self, element: _Element) -> list[_Block]:
rows: list[list[str]] = []
def collect(node: _Element) -> None:
for child in node.children:
if isinstance(child, str) or child.tag in _DROPPED or child.tag == "table":
continue
if child.tag == "tr":
cells = [
_finish_inline(self.inline_children(cell)).replace("\n", " ").replace("|", "\\|")
for cell in child.children
if isinstance(cell, _Element) and cell.tag in ("td", "th")
]
if any(cells):
rows.append(cells)
else:
collect(child)
collect(element)
if not rows:
return []
width = max(len(row) for row in rows)
padded = [row + [""] * (width - len(row)) for row in rows]
lines = ["| " + " | ".join(padded[0]) + " |", "|" + " --- |" * width]
lines += ["| " + " | ".join(row) + " |" for row in padded[1:]]
return [_Block("text", "\n".join(lines))]
# Inline -------------------------------------------------------------
def inline_children(self, element: _Element) -> str:
parts = []
for child in element.children:
if isinstance(child, str):
parts.append(child)
elif child.tag in _DROPPED:
continue
else:
part = self.inline(child)
# A block element met in inline context still separates words.
parts.append(f" {part} " if child.tag in _BLOCK else part)
return "".join(parts)
def inline(self, element: _Element) -> str:
tag = element.tag
if tag in _DROPPED:
return ""
if tag == "br":
return _BR
if tag == "img":
alt = _WS.sub(" ", element.attrs.get("alt", "")).strip()
src = element.attrs.get("src", "").strip()
if not src or src.lower().startswith("data:"):
return alt
return f"![{alt}]({self.url(src)})"
if tag == "code":
text = _WS.sub(" ", element.text()).strip()
if not text:
return ""
longest = max((len(run) for run in re.findall(r"`+", text)), default=0)
ticks = "`" * (longest + 1)
pad = " " if text.startswith("`") or text.endswith("`") else ""
return f"{ticks}{pad}{text}{pad}{ticks}"
if tag == "a":
text = _WS.sub(" ", self.inline_children(element)).strip()
href = element.attrs.get("href", "").strip()
if not href or href.startswith("#") or href.lower().startswith("javascript:"):
return text
if not text:
return ""
return f"[{text}]({self.url(href)})"
return self.inline_children(element)
def _finish_inline(text: str) -> str:
text = _WS.sub(" ", text)
lines = [line.strip() for line in text.split(_BR)]
return "\n".join(lines).strip("\n").strip()
def _join(blocks: list[_Block]) -> str:
return "\n\n".join(block.text for block in blocks)
def derive_text(html: str, base_url: str) -> str:
"""The reading text of an HTML document: the content root's blocks as
Markdown-like text, with links made absolute against `base_url`."""
document = parse(html)
renderer = _Renderer(base_url)
root = content_root(document)
blocks = renderer.block(root) if root.tag != "#document" else renderer.blocks(root)
return _join(blocks)
# --- the capture file -------------------------------------------------------
def utc_now() -> str:
return datetime.datetime.now(datetime.timezone.utc).replace(microsecond=0).strftime("%Y-%m-%dT%H:%M:%SZ")
_PLAIN_SCALAR = re.compile(r"[^\s\-?:,\[\]{}#&*!|>'\"%@`][^\n]*")
def _scalar(value: str) -> str:
"""A YAML scalar for one header value: plain where YAML reads it back
unchanged, double-quoted (a JSON string is valid YAML) otherwise - a page
title with `: ` in it would end the block's parseability."""
if value and _PLAIN_SCALAR.fullmatch(value) and ": " not in value and " #" not in value and value == value.strip():
return value
return json.dumps(value, ensure_ascii=False)
def header(fields: list[tuple[str, str]]) -> str:
lines = ["---"] + [f"{key}: {_scalar(value)}" for key, value in fields] + ["---"]
return "\n".join(lines) + "\n"
@dataclass(frozen=True)
class Derivation:
markdown: bytes # header + derived text, UTF-8, LF
text_chars: int
title: str
decoded: Decoded
def derive_document(
body: bytes,
base_url: str,
html_name: str,
*,
response: Optional[Response] = None,
url: Optional[str] = None,
derived_at: Optional[str] = None,
) -> Derivation:
"""The `.md` beside a captured `.html`: the fixed header, then the derived
text. With `response` it describes a fetch; without, a page a human saved
(`url` and `derived_at` then fill the header instead)."""
decoded = decode(body, response.content_type if response else None)
document_title = title_of(parse(decoded.text))
text = derive_text(decoded.text, base_url)
charset = f"{decoded.charset} (from: {decoded.source})"
if response is not None:
fields = [
("fetched_by", "wikitool raw fetch"),
("url", response.url),
("final_url", response.final_url),
("retrieved", derived_at or utc_now()),
("http_status", str(response.status)),
("content_type", response.content_type),
("charset", charset),
("title", document_title),
("derived_from", html_name),
]
else:
fields = [
("fetched_by", "wikitool raw fetch --html"),
("url", url or ""),
("derived", derived_at or utc_now()),
("charset", charset),
("title", document_title),
("derived_from", html_name),
]
content = header(fields) + "\n" + (text + "\n" if text else "")
return Derivation(content.encode("utf-8"), len(text), document_title, decoded)
# --- naming -----------------------------------------------------------------
def slug(value: str) -> str:
value = unicodedata.normalize("NFKD", value).encode("ascii", "ignore").decode("ascii").lower()
value = re.sub(r"[^a-z0-9]+", "-", value).strip("-")
return value[:STEM_MAX_CHARS].rstrip("-")
def stem_for(url: str) -> str:
"""The last non-empty path segment without its extension, else the host
name - ASCII, lowercase, `-`-separated, at most 60 characters."""
parts = urllib.parse.urlsplit(url)
segments = [s for s in urllib.parse.unquote(parts.path).split("/") if s]
if segments:
last = segments[-1]
base = last.rsplit(".", 1)[0] if "." in last.strip(".") else last
candidate = slug(base)
if candidate:
return candidate
return slug(parts.hostname or "") or "page"