Files changed: - CHANGES.md - README.md - VERSION - instructions/wiki-ingest/SKILL.md - raw/CONTRACT.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/cli_contract.py - tools/chemenu/commands/raw_cmd.py - tools/chemenu/tests/test_cli.py - tools/chemenu/tests/test_portability.py - tools/chemenu/tests/test_raw_fetch.py - tools/chemenu/web_capture.py Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
650 lines
24 KiB
Python
650 lines
24 KiB
Python
"""Capturing a web page for `raw/`: fetch the bytes, decide their character
|
|
set, and derive a Markdown-like reading text from them - with no CLI attached.
|
|
|
|
`wikitool raw fetch` is the terminal adapter over this module; it decides where
|
|
the files go and what the caller is told. Everything here is deterministic on
|
|
purpose: the same bytes always produce the same derived text, so two sessions
|
|
capturing the same article produce the same `raw/` bundle rather than two
|
|
hand-built variants of it. That is also why there is no readability heuristic
|
|
(text density, line numbers): a heuristic tuned once drifts the next time it is
|
|
tuned, and the received HTML is kept beside the derivation anyway, so what the
|
|
derivation drops is never lost.
|
|
|
|
Standard library only. A third-party HTML library would be a new entry in
|
|
`tools/requirements.txt`, missing from an instance's venv until someone runs
|
|
`pip install` after an upgrade - which would make this a breaking change for a
|
|
gain the kept HTML already covers.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import codecs
|
|
import datetime
|
|
import email.message
|
|
import json
|
|
import mimetypes
|
|
import re
|
|
import time
|
|
import unicodedata
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
from dataclasses import dataclass
|
|
from html.parser import HTMLParser
|
|
from typing import Optional, Union
|
|
|
|
from chemenu.errors import BackendError, ValidationError
|
|
|
|
TIMEOUT_SECONDS = 30.0
|
|
MAX_BYTES = 25 * 1024 * 1024
|
|
SHORT_TEXT_CHARS = 200
|
|
STEM_MAX_CHARS = 60
|
|
META_SCAN_BYTES = 4096
|
|
|
|
ALLOWED_SCHEMES = ("http", "https")
|
|
HTML_TYPES = ("text/html", "application/xhtml+xml")
|
|
|
|
# Content types stored as received, under a fixed extension, without a
|
|
# derivation: they are already text a session can read and cite directly.
|
|
PLAIN_TEXT_EXTENSIONS = {"text/plain": ".txt", "text/markdown": ".md"}
|
|
|
|
|
|
# --- fetching ---------------------------------------------------------------
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Response:
|
|
url: str
|
|
final_url: str
|
|
status: int
|
|
content_type: str
|
|
body: bytes
|
|
|
|
@property
|
|
def media_type(self) -> str:
|
|
"""The bare media type, lowercased - `text/html` out of
|
|
`text/html; charset=utf-8`; empty when the server sent none."""
|
|
return media_type(self.content_type)
|
|
|
|
|
|
def media_type(content_type: str) -> str:
|
|
if not content_type.strip():
|
|
return ""
|
|
message = email.message.Message()
|
|
message["content-type"] = content_type
|
|
return message.get_content_type().lower()
|
|
|
|
|
|
def check_url(url: str) -> None:
|
|
"""Only `http`/`https`. A `file:` URL would be a way into `incoming/` that
|
|
bypasses a human dropping the file there; `ftp:` and the rest are not
|
|
pages."""
|
|
parts = urllib.parse.urlsplit(url)
|
|
if parts.scheme.lower() not in ALLOWED_SCHEMES or not parts.netloc:
|
|
raise ValidationError(
|
|
f"{url} is not an http(s) URL - `raw fetch` fetches web pages only. A local file "
|
|
"goes into incoming/ by hand."
|
|
)
|
|
|
|
|
|
class _HttpOnlyRedirects(urllib.request.HTTPRedirectHandler):
|
|
"""urllib follows a redirect to `ftp:` on its own; the scheme rule above
|
|
has to hold for every hop, not only for the URL the caller typed."""
|
|
|
|
def redirect_request(self, req, fp, code, msg, headers, newurl): # noqa: D102 - urllib's hook
|
|
check_url(urllib.parse.urljoin(req.full_url, newurl))
|
|
return super().redirect_request(req, fp, code, msg, headers, newurl)
|
|
|
|
|
|
def fetch(
|
|
url: str,
|
|
user_agent: str,
|
|
timeout: float = TIMEOUT_SECONDS,
|
|
max_bytes: int = MAX_BYTES,
|
|
) -> Response:
|
|
"""GET `url` and return the body exactly as received.
|
|
|
|
No cookies (the opener carries no cookie processor) and no content
|
|
decoding - urllib asks for `identity`, so the bytes are the page, not a
|
|
compressed transfer of it. `timeout` bounds the whole transfer, not only a
|
|
single read, so a server that trickles bytes cannot hold the call open.
|
|
Raises `ValidationError` for a URL outside `http`/`https` (also on a
|
|
redirect) and `BackendError` for everything the network does: an HTTP
|
|
error status, an unreachable host, a timeout, a body over `max_bytes`.
|
|
"""
|
|
check_url(url)
|
|
opener = urllib.request.build_opener(_HttpOnlyRedirects)
|
|
request = urllib.request.Request(url, headers={"User-Agent": user_agent, "Accept": "*/*"})
|
|
deadline = time.monotonic() + timeout
|
|
try:
|
|
with opener.open(request, timeout=timeout) as response:
|
|
declared = response.headers.get("Content-Length")
|
|
if declared and declared.strip().isdigit() and int(declared) > max_bytes:
|
|
raise BackendError(_too_large(url, max_bytes))
|
|
chunks: list[bytes] = []
|
|
size = 0
|
|
while True:
|
|
if time.monotonic() > deadline:
|
|
raise BackendError(f"{url} did not finish within {timeout:g} s.")
|
|
chunk = response.read(64 * 1024)
|
|
if not chunk:
|
|
break
|
|
size += len(chunk)
|
|
if size > max_bytes:
|
|
raise BackendError(_too_large(url, max_bytes))
|
|
chunks.append(chunk)
|
|
return Response(
|
|
url=url,
|
|
final_url=response.geturl(),
|
|
status=response.status,
|
|
content_type=response.headers.get("Content-Type", "") or "",
|
|
body=b"".join(chunks),
|
|
)
|
|
except urllib.error.HTTPError as exc:
|
|
raise BackendError(f"{url} answered HTTP {exc.code} {exc.reason}.") from exc
|
|
except urllib.error.URLError as exc:
|
|
if isinstance(exc.reason, TimeoutError):
|
|
raise BackendError(f"{url} did not answer within {timeout:g} s.") from exc
|
|
raise BackendError(f"Could not reach {url}: {exc.reason}") from exc
|
|
except TimeoutError as exc:
|
|
raise BackendError(f"{url} did not answer within {timeout:g} s.") from exc
|
|
except OSError as exc:
|
|
raise BackendError(f"Could not fetch {url}: {exc}") from exc
|
|
|
|
|
|
def _too_large(url: str, max_bytes: int) -> str:
|
|
return f"{url} is larger than {max_bytes // (1024 * 1024)} MiB - nothing was written."
|
|
|
|
|
|
def extension_for(content_type: str) -> str:
|
|
"""The file extension a non-HTML response is stored under. Read from
|
|
Python's built-in table only - `mimetypes.MimeTypes()` ignores the
|
|
machine's `/etc/mime.types`, so the answer does not depend on the host."""
|
|
kind = media_type(content_type)
|
|
if kind in PLAIN_TEXT_EXTENSIONS:
|
|
return PLAIN_TEXT_EXTENSIONS[kind]
|
|
if not kind:
|
|
return ".bin"
|
|
return mimetypes.MimeTypes().guess_extension(kind) or ".bin"
|
|
|
|
|
|
# --- character set ----------------------------------------------------------
|
|
|
|
_BOMS = (
|
|
(codecs.BOM_UTF8, "utf-8"),
|
|
(codecs.BOM_UTF32_LE, "utf-32-le"),
|
|
(codecs.BOM_UTF32_BE, "utf-32-be"),
|
|
(codecs.BOM_UTF16_LE, "utf-16-le"),
|
|
(codecs.BOM_UTF16_BE, "utf-16-be"),
|
|
)
|
|
|
|
# Covers both `<meta charset="x">` and
|
|
# `<meta http-equiv="Content-Type" content="text/html; charset=x">`.
|
|
_META_CHARSET = re.compile(rb"""<meta\b[^>]*?charset\s*=\s*["']?\s*([A-Za-z0-9_.:\-]+)""", re.IGNORECASE)
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Decoded:
|
|
text: str
|
|
charset: str
|
|
source: str # bom | header | meta | default
|
|
replaced: int # how many U+FFFD the decoding introduced
|
|
|
|
|
|
def _known(label: Optional[str]) -> Optional[str]:
|
|
if not label:
|
|
return None
|
|
label = label.strip().lower()
|
|
try:
|
|
codecs.lookup(label)
|
|
except LookupError:
|
|
return None
|
|
return label
|
|
|
|
|
|
def decode(body: bytes, content_type: Optional[str]) -> Decoded:
|
|
"""Decode an HTML body, in a fixed order: byte-order mark, then the
|
|
`charset` of the HTTP `Content-Type` (`None` when there was no HTTP
|
|
response, as for a page saved from a browser), then a `<meta>` declaration
|
|
in the first 4 KiB, then UTF-8. A label Python does not know is skipped
|
|
like an absent one. Undecodable bytes are replaced, never dropped, and
|
|
counted, so the caller can say so."""
|
|
charset, source, start = None, "", 0
|
|
for bom, name in _BOMS:
|
|
if body.startswith(bom):
|
|
charset, source, start = name, "bom", len(bom)
|
|
break
|
|
if charset is None and content_type:
|
|
message = email.message.Message()
|
|
message["content-type"] = content_type
|
|
charset = _known(message.get_content_charset())
|
|
source = "header" if charset else ""
|
|
if charset is None:
|
|
match = _META_CHARSET.search(body[:META_SCAN_BYTES])
|
|
charset = _known(match.group(1).decode("ascii", "replace")) if match else None
|
|
source = "meta" if charset else ""
|
|
if charset is None:
|
|
charset, source = "utf-8", "default"
|
|
payload = body[start:]
|
|
try:
|
|
return Decoded(payload.decode(charset), charset, source, 0)
|
|
except UnicodeDecodeError:
|
|
text = payload.decode(charset, errors="replace")
|
|
before = payload.decode(charset, errors="ignore").count("�")
|
|
return Decoded(text, charset, source, text.count("�") - before)
|
|
|
|
|
|
# --- HTML -> text -----------------------------------------------------------
|
|
|
|
_VOID = frozenset(
|
|
"area base br col embed hr img input keygen link meta param source track wbr".split()
|
|
)
|
|
_DROPPED = frozenset(
|
|
"script style noscript nav header footer aside form template svg head title".split()
|
|
)
|
|
_BLOCK = frozenset(
|
|
"""address article blockquote body center dd details dialog div dl dt fieldset
|
|
figcaption figure h1 h2 h3 h4 h5 h6 hgroup hr html li main ol p pre section summary
|
|
table tbody thead tfoot tr td th caption ul""".split()
|
|
)
|
|
# An open `<p>` ends where one of these starts - the parser's share of the
|
|
# implied end tags real-world HTML relies on.
|
|
_CLOSES_P = frozenset(
|
|
"""address article blockquote div dl fieldset figure h1 h2 h3 h4 h5 h6 hr main ol p pre
|
|
section table ul""".split()
|
|
)
|
|
_HEADINGS = {f"h{n}": n for n in range(1, 7)}
|
|
_WS = re.compile(r"[ \t\n\r\f\v]+")
|
|
_BR = "\x00"
|
|
|
|
|
|
class _Element:
|
|
__slots__ = ("tag", "attrs", "children", "parent")
|
|
|
|
def __init__(self, tag: str, attrs: dict[str, str], parent: Optional["_Element"]):
|
|
self.tag = tag
|
|
self.attrs = attrs
|
|
self.children: list[Union["_Element", str]] = []
|
|
self.parent = parent
|
|
|
|
def iter(self):
|
|
yield self
|
|
for child in self.children:
|
|
if isinstance(child, _Element):
|
|
yield from child.iter()
|
|
|
|
def text(self) -> str:
|
|
return "".join(c if isinstance(c, str) else c.text() for c in self.children)
|
|
|
|
|
|
class _TreeBuilder(HTMLParser):
|
|
"""A forgiving tree: an end tag closes the nearest open element of that
|
|
name and is ignored when none is open, void elements never take children,
|
|
and `<p>`/`<li>`/`<dt>`/`<dd>`/`<tr>`/`<td>`/`<th>`/`<option>` end where
|
|
the next sibling of their kind begins."""
|
|
|
|
def __init__(self) -> None:
|
|
super().__init__(convert_charrefs=True)
|
|
self.root = _Element("#document", {}, None)
|
|
self.stack = [self.root]
|
|
|
|
def _close_to(self, tag: str, stop_at: tuple[str, ...] = ()) -> None:
|
|
for index in range(len(self.stack) - 1, 0, -1):
|
|
name = self.stack[index].tag
|
|
if name == tag:
|
|
del self.stack[index:]
|
|
return
|
|
if name in stop_at:
|
|
return
|
|
|
|
def handle_starttag(self, tag, attrs):
|
|
if tag in _CLOSES_P:
|
|
self._close_to("p", stop_at=("div", "li", "td", "th", "blockquote", "section", "article", "main", "body"))
|
|
if tag == "li":
|
|
self._close_to("li", stop_at=("ul", "ol"))
|
|
elif tag in ("dt", "dd"):
|
|
self._close_to("dt", stop_at=("dl",))
|
|
self._close_to("dd", stop_at=("dl",))
|
|
elif tag == "tr":
|
|
self._close_to("tr", stop_at=("table",))
|
|
elif tag in ("td", "th"):
|
|
self._close_to("td", stop_at=("tr", "table"))
|
|
self._close_to("th", stop_at=("tr", "table"))
|
|
elif tag == "option":
|
|
self._close_to("option", stop_at=("select",))
|
|
element = _Element(tag, {k: (v or "") for k, v in attrs}, self.stack[-1])
|
|
self.stack[-1].children.append(element)
|
|
if tag not in _VOID:
|
|
self.stack.append(element)
|
|
|
|
def handle_startendtag(self, tag, attrs):
|
|
element = _Element(tag, {k: (v or "") for k, v in attrs}, self.stack[-1])
|
|
self.stack[-1].children.append(element)
|
|
|
|
def handle_endtag(self, tag):
|
|
if tag not in _VOID:
|
|
self._close_to(tag)
|
|
|
|
def handle_data(self, data):
|
|
self.stack[-1].children.append(data)
|
|
|
|
|
|
def parse(text: str) -> _Element:
|
|
builder = _TreeBuilder()
|
|
builder.feed(text)
|
|
builder.close()
|
|
return builder.root
|
|
|
|
|
|
def title_of(document: _Element) -> str:
|
|
for element in document.iter():
|
|
if element.tag == "title":
|
|
return _WS.sub(" ", element.text()).strip()
|
|
return ""
|
|
|
|
|
|
def content_root(document: _Element) -> _Element:
|
|
"""`<main>`, else the one `<article>` if there is exactly one, else
|
|
`<body>`, else the whole document."""
|
|
elements = list(document.iter())
|
|
for element in elements:
|
|
if element.tag == "main":
|
|
return element
|
|
articles = [e for e in elements if e.tag == "article"]
|
|
if len(articles) == 1:
|
|
return articles[0]
|
|
for element in elements:
|
|
if element.tag == "body":
|
|
return element
|
|
return document
|
|
|
|
|
|
@dataclass
|
|
class _Block:
|
|
kind: str # "text" or "list"
|
|
text: str
|
|
|
|
|
|
class _Renderer:
|
|
def __init__(self, base_url: str) -> None:
|
|
self.base_url = base_url
|
|
|
|
def url(self, href: str) -> str:
|
|
return urllib.parse.urljoin(self.base_url, href.strip())
|
|
|
|
# Blocks -------------------------------------------------------------
|
|
|
|
def blocks(self, element: _Element) -> list[_Block]:
|
|
"""The blocks of a container: runs of inline content become one
|
|
paragraph each, block children contribute their own blocks."""
|
|
out: list[_Block] = []
|
|
inline: list[str] = []
|
|
|
|
def flush() -> None:
|
|
text = _finish_inline("".join(inline))
|
|
if text:
|
|
out.append(_Block("text", text))
|
|
inline.clear()
|
|
|
|
for child in element.children:
|
|
if isinstance(child, str):
|
|
inline.append(child)
|
|
elif child.tag in _DROPPED:
|
|
continue
|
|
elif child.tag in _BLOCK:
|
|
flush()
|
|
out.extend(self.block(child))
|
|
else:
|
|
inline.append(self.inline(child))
|
|
flush()
|
|
return out
|
|
|
|
def block(self, element: _Element) -> list[_Block]:
|
|
tag = element.tag
|
|
if tag in _HEADINGS:
|
|
text = _finish_inline(self.inline_children(element))
|
|
return [_Block("text", "#" * _HEADINGS[tag] + " " + text)] if text else []
|
|
if tag == "p":
|
|
text = _finish_inline(self.inline_children(element))
|
|
return [_Block("text", text)] if text else []
|
|
if tag == "pre":
|
|
return self.pre(element)
|
|
if tag in ("ul", "ol"):
|
|
return self.list(element)
|
|
if tag == "blockquote":
|
|
inner = _join(self.blocks(element))
|
|
if not inner:
|
|
return []
|
|
return [_Block("text", "\n".join(f"> {line}" if line else ">" for line in inner.split("\n")))]
|
|
if tag == "hr":
|
|
return [_Block("text", "---")]
|
|
if tag == "table":
|
|
return self.table(element)
|
|
return self.blocks(element)
|
|
|
|
def pre(self, element: _Element) -> list[_Block]:
|
|
text = element.text().strip("\n")
|
|
if not text.strip():
|
|
return []
|
|
longest = max((len(run) for run in re.findall(r"`+", text)), default=0)
|
|
fence = "`" * max(3, longest + 1)
|
|
return [_Block("text", f"{fence}\n{text}\n{fence}")]
|
|
|
|
def list(self, element: _Element) -> list[_Block]:
|
|
ordered = element.tag == "ol"
|
|
start = element.attrs.get("start", "").strip()
|
|
number = int(start) if ordered and start.lstrip("-").isdigit() else 1
|
|
lines: list[str] = []
|
|
for child in element.children:
|
|
if isinstance(child, str) or child.tag in _DROPPED:
|
|
continue
|
|
if child.tag == "li":
|
|
content = self.blocks(child)
|
|
elif child.tag in ("ul", "ol"):
|
|
# A list nested directly in a list, without its own <li>.
|
|
content = self.block(child)
|
|
else:
|
|
continue
|
|
if not content:
|
|
continue
|
|
marker = f"{number}. " if ordered else "- "
|
|
number += 1
|
|
body = ""
|
|
for index, block in enumerate(content):
|
|
if index:
|
|
body += "\n" if block.kind == "list" else "\n\n"
|
|
body += block.text
|
|
indent = " " * len(marker)
|
|
lines.append("\n".join(
|
|
(marker if i == 0 else (indent if line else "")) + line
|
|
for i, line in enumerate(body.split("\n"))
|
|
))
|
|
return [_Block("list", "\n".join(lines))] if lines else []
|
|
|
|
def table(self, element: _Element) -> list[_Block]:
|
|
rows: list[list[str]] = []
|
|
|
|
def collect(node: _Element) -> None:
|
|
for child in node.children:
|
|
if isinstance(child, str) or child.tag in _DROPPED or child.tag == "table":
|
|
continue
|
|
if child.tag == "tr":
|
|
cells = [
|
|
_finish_inline(self.inline_children(cell)).replace("\n", " ").replace("|", "\\|")
|
|
for cell in child.children
|
|
if isinstance(cell, _Element) and cell.tag in ("td", "th")
|
|
]
|
|
if any(cells):
|
|
rows.append(cells)
|
|
else:
|
|
collect(child)
|
|
|
|
collect(element)
|
|
if not rows:
|
|
return []
|
|
width = max(len(row) for row in rows)
|
|
padded = [row + [""] * (width - len(row)) for row in rows]
|
|
lines = ["| " + " | ".join(padded[0]) + " |", "|" + " --- |" * width]
|
|
lines += ["| " + " | ".join(row) + " |" for row in padded[1:]]
|
|
return [_Block("text", "\n".join(lines))]
|
|
|
|
# Inline -------------------------------------------------------------
|
|
|
|
def inline_children(self, element: _Element) -> str:
|
|
parts = []
|
|
for child in element.children:
|
|
if isinstance(child, str):
|
|
parts.append(child)
|
|
elif child.tag in _DROPPED:
|
|
continue
|
|
else:
|
|
part = self.inline(child)
|
|
# A block element met in inline context still separates words.
|
|
parts.append(f" {part} " if child.tag in _BLOCK else part)
|
|
return "".join(parts)
|
|
|
|
def inline(self, element: _Element) -> str:
|
|
tag = element.tag
|
|
if tag in _DROPPED:
|
|
return ""
|
|
if tag == "br":
|
|
return _BR
|
|
if tag == "img":
|
|
alt = _WS.sub(" ", element.attrs.get("alt", "")).strip()
|
|
src = element.attrs.get("src", "").strip()
|
|
if not src or src.lower().startswith("data:"):
|
|
return alt
|
|
return f"})"
|
|
if tag == "code":
|
|
text = _WS.sub(" ", element.text()).strip()
|
|
if not text:
|
|
return ""
|
|
longest = max((len(run) for run in re.findall(r"`+", text)), default=0)
|
|
ticks = "`" * (longest + 1)
|
|
pad = " " if text.startswith("`") or text.endswith("`") else ""
|
|
return f"{ticks}{pad}{text}{pad}{ticks}"
|
|
if tag == "a":
|
|
text = _WS.sub(" ", self.inline_children(element)).strip()
|
|
href = element.attrs.get("href", "").strip()
|
|
if not href or href.startswith("#") or href.lower().startswith("javascript:"):
|
|
return text
|
|
if not text:
|
|
return ""
|
|
return f"[{text}]({self.url(href)})"
|
|
return self.inline_children(element)
|
|
|
|
|
|
def _finish_inline(text: str) -> str:
|
|
text = _WS.sub(" ", text)
|
|
lines = [line.strip() for line in text.split(_BR)]
|
|
return "\n".join(lines).strip("\n").strip()
|
|
|
|
|
|
def _join(blocks: list[_Block]) -> str:
|
|
return "\n\n".join(block.text for block in blocks)
|
|
|
|
|
|
def derive_text(html: str, base_url: str) -> str:
|
|
"""The reading text of an HTML document: the content root's blocks as
|
|
Markdown-like text, with links made absolute against `base_url`."""
|
|
document = parse(html)
|
|
renderer = _Renderer(base_url)
|
|
root = content_root(document)
|
|
blocks = renderer.block(root) if root.tag != "#document" else renderer.blocks(root)
|
|
return _join(blocks)
|
|
|
|
|
|
# --- the capture file -------------------------------------------------------
|
|
|
|
|
|
def utc_now() -> str:
|
|
return datetime.datetime.now(datetime.timezone.utc).replace(microsecond=0).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
|
|
|
|
_PLAIN_SCALAR = re.compile(r"[^\s\-?:,\[\]{}#&*!|>'\"%@`][^\n]*")
|
|
|
|
|
|
def _scalar(value: str) -> str:
|
|
"""A YAML scalar for one header value: plain where YAML reads it back
|
|
unchanged, double-quoted (a JSON string is valid YAML) otherwise - a page
|
|
title with `: ` in it would end the block's parseability."""
|
|
if value and _PLAIN_SCALAR.fullmatch(value) and ": " not in value and " #" not in value and value == value.strip():
|
|
return value
|
|
return json.dumps(value, ensure_ascii=False)
|
|
|
|
|
|
def header(fields: list[tuple[str, str]]) -> str:
|
|
lines = ["---"] + [f"{key}: {_scalar(value)}" for key, value in fields] + ["---"]
|
|
return "\n".join(lines) + "\n"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Derivation:
|
|
markdown: bytes # header + derived text, UTF-8, LF
|
|
text_chars: int
|
|
title: str
|
|
decoded: Decoded
|
|
|
|
|
|
def derive_document(
|
|
body: bytes,
|
|
base_url: str,
|
|
html_name: str,
|
|
*,
|
|
response: Optional[Response] = None,
|
|
url: Optional[str] = None,
|
|
derived_at: Optional[str] = None,
|
|
) -> Derivation:
|
|
"""The `.md` beside a captured `.html`: the fixed header, then the derived
|
|
text. With `response` it describes a fetch; without, a page a human saved
|
|
(`url` and `derived_at` then fill the header instead)."""
|
|
decoded = decode(body, response.content_type if response else None)
|
|
document_title = title_of(parse(decoded.text))
|
|
text = derive_text(decoded.text, base_url)
|
|
charset = f"{decoded.charset} (from: {decoded.source})"
|
|
if response is not None:
|
|
fields = [
|
|
("fetched_by", "wikitool raw fetch"),
|
|
("url", response.url),
|
|
("final_url", response.final_url),
|
|
("retrieved", derived_at or utc_now()),
|
|
("http_status", str(response.status)),
|
|
("content_type", response.content_type),
|
|
("charset", charset),
|
|
("title", document_title),
|
|
("derived_from", html_name),
|
|
]
|
|
else:
|
|
fields = [
|
|
("fetched_by", "wikitool raw fetch --html"),
|
|
("url", url or ""),
|
|
("derived", derived_at or utc_now()),
|
|
("charset", charset),
|
|
("title", document_title),
|
|
("derived_from", html_name),
|
|
]
|
|
content = header(fields) + "\n" + (text + "\n" if text else "")
|
|
return Derivation(content.encode("utf-8"), len(text), document_title, decoded)
|
|
|
|
|
|
# --- naming -----------------------------------------------------------------
|
|
|
|
|
|
def slug(value: str) -> str:
|
|
value = unicodedata.normalize("NFKD", value).encode("ascii", "ignore").decode("ascii").lower()
|
|
value = re.sub(r"[^a-z0-9]+", "-", value).strip("-")
|
|
return value[:STEM_MAX_CHARS].rstrip("-")
|
|
|
|
|
|
def stem_for(url: str) -> str:
|
|
"""The last non-empty path segment without its extension, else the host
|
|
name - ASCII, lowercase, `-`-separated, at most 60 characters."""
|
|
parts = urllib.parse.urlsplit(url)
|
|
segments = [s for s in urllib.parse.unquote(parts.path).split("/") if s]
|
|
if segments:
|
|
last = segments[-1]
|
|
base = last.rsplit(".", 1)[0] if "." in last.strip(".") else last
|
|
candidate = slug(base)
|
|
if candidate:
|
|
return candidate
|
|
return slug(parts.hostname or "") or "page"
|