feat: raw fetch - a sanctioned intake for a URL into incoming/, HTML as received plus derived text (#120)
CI / verify (push) Successful in 5m21s
CI / pwsh (push) Successful in 2m6s
Release / release (push) Successful in 34s

Files changed:
- CHANGES.md
- README.md
- VERSION
- instructions/wiki-ingest/SKILL.md
- raw/CONTRACT.md
- tools/CONTRACT.md
- tools/README.md
- tools/chemenu/cli_contract.py
- tools/chemenu/commands/raw_cmd.py
- tools/chemenu/tests/test_cli.py
- tools/chemenu/tests/test_portability.py
- tools/chemenu/tests/test_raw_fetch.py
- tools/chemenu/web_capture.py

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
This commit is contained in:
torbenandClaude Opus 5.5 committed 2026-10-02 22:26:06 +02:00
1 parent c0f324ff96
commit fba263af68
13 files changed
+1571 -16

No files matched your search

+265 -4
View File
@@ -70,15 +70,18 @@ from pathlib import Path
from typing import Optional
import typer
from rich.markup import escape
from chemenu import cli_contract, config
from chemenu import cli_contract, config, web_capture
from chemenu.commands._util import check_path_budget, fail, rel_path, success
from chemenu.errors import ChemenuError
from chemenu.frontmatter_io import write_page
from chemenu.kb_scan import load_kb_pages
from chemenu.provenance import citing_pages, source_pages_by_raw_file, source_raw_files
from chemenu.type_resolver import resolver
from chemenu.version import VersionError, read_version
app = typer.Typer(help="Promote raw material out of incoming/ into raw/.")
app = typer.Typer(help="Bring raw material into incoming/, and promote it from there into raw/.")
# The type this command always promotes into eventually - hardcoded rather
# than derived from --page, because there is no page yet in the common case
@@ -118,8 +121,8 @@ def _validate_under_incoming(path: Path, incoming: Path) -> None:
rel = path.relative_to(incoming)
except ValueError:
fail(
f"{rel_path(path)} is not under incoming/ - `raw accept` only promotes files "
"from there. See raw/CONTRACT.md."
f"{rel_path(path)} is not under incoming/ - `raw accept` and `raw fetch --html` "
"only take files from there. See raw/CONTRACT.md."
)
if len(rel.parts) < 1:
fail(f"incoming/{rel.as_posix()} names no file.")
@@ -658,3 +661,261 @@ def raw_accept_command(
f" --set fidelity={fidelity} --set authority={authority} \\\n"
" --set source_type=<category>"
)
# --- raw fetch ----------------------------------------------------------------
#
# The sanctioned intake for a URL the user names. It ends in `incoming/`, never
# in `raw/`: promoting stays `raw accept`'s job, so `wiki-ingest` keeps its
# order (the commitment question comes before the promotion) and the capture
# fields are asked exactly once, there. The fetch and derivation themselves live
# in `chemenu.web_capture`; this adapter decides file names and what the caller
# is told.
_TEASER_HINT = (
" If the text stops at a teaser (paywall, login, script-rendered page): save the page from a\n"
" logged-in browser to incoming/ and run tools/wikitool raw fetch --html "
"incoming/<file>.html --url {url}"
)
def _user_agent() -> str:
try:
version = str(read_version())
except VersionError:
version = "unknown"
return f"chemenu-wikitool/{version} (raw fetch)"
def _next_lines(files: list[Path], source_url: str) -> str:
paths = " ".join(rel_path(f) for f in files)
return (
" Next (after the commitment question in wiki-ingest):\n"
f" tools/wikitool raw accept --fidelity published --authority <value> {paths}\n"
f" and on the source page: --set source_url={source_url}"
)
def _refuse_existing(targets: list[Path]) -> None:
existing = [t for t in targets if t.exists()]
if existing:
fail(escape(
f"{', '.join(rel_path(t) for t in existing)} already exists in incoming/ - raw fetch "
"never overwrites. Accept or remove the file that is there, or pass --name <stem>."
))
def _write_exclusive(writes: list[tuple[Path, bytes]]) -> None:
"""Create every file or none: `x` mode refuses a file that appeared since
the check above, and a failure part-way removes what this call wrote."""
written: list[Path] = []
try:
for path, data in writes:
path.parent.mkdir(parents=True, exist_ok=True)
with path.open("xb") as handle:
written.append(path)
handle.write(data)
except OSError as exc:
for path in written:
path.unlink(missing_ok=True)
fail(escape(f"Could not write {rel_path(path)}: {exc}. Nothing was left in incoming/."))
def _warn(derivation: web_capture.Derivation, md_path: Path) -> None:
if derivation.decoded.replaced:
typer.echo(
f"WARN {derivation.decoded.replaced} undecodable byte(s) under "
f"{derivation.decoded.charset} were replaced with U+FFFD in {rel_path(md_path)} - "
"the .html beside it keeps the original bytes."
)
if derivation.text_chars < web_capture.SHORT_TEXT_CHARS:
typer.echo(
f"WARN The derived text is only {derivation.text_chars} characters - likely a "
"script-rendered page, a login wall or an empty response. Read "
f"{rel_path(md_path)} before accepting it."
)
def _fetch_html_file(html: Path, source_url: Optional[str]) -> None:
if source_url is None:
fail("--html needs --url <url>: it goes into the header, resolves relative links, and "
"becomes source_url on the source page.")
try:
web_capture.check_url(source_url)
except ChemenuError as exc:
fail(escape(str(exc)))
path = _resolve(html)
_validate_under_incoming(path, _incoming_dir())
if not path.is_file():
fail(f"{rel_path(path)} does not exist or is not a file.")
md_path = path.with_suffix(".md")
if md_path == path:
fail(f"{rel_path(path)} is already a .md file - --html takes the saved HTML page.")
_refuse_existing([md_path])
derivation = web_capture.derive_document(path.read_bytes(), source_url, path.name, url=source_url)
_write_exclusive([(md_path, derivation.markdown)])
_warn(derivation, md_path)
success(escape(
f"Derived {rel_path(md_path)} from {rel_path(path)} (no network access).\n"
+ _next_lines([path, md_path], source_url)
))
@cli_contract.record(cli_contract.CommandRecord(
path="raw fetch",
summary="Capture a web page the user names into `incoming/`: the HTML as received plus a "
"derived text, for `raw accept` to promote.",
synopsis=(
cli_contract.Variant(
usage="raw fetch <url> [--name <stem>]",
notes="Fetch the page and write `incoming/<stem>.html` and `incoming/<stem>.md`",
),
cli_contract.Variant(
usage="raw fetch --html incoming/<file>.html --url <url>",
notes="Derive the `.md` from a page a human saved from their browser - no network "
"access",
),
),
properties=cli_contract.Properties(
effect=cli_contract.Effect.WRITE,
idempotent=cli_contract.Idempotent.NO,
atomic="Yes for what it leaves behind - one or two new files in `incoming/`, created "
"exclusively; a failure on the second removes the first",
budget=cli_contract.Budget.COUNTED,
network=cli_contract.Network.YES,
),
notes=(
"Writes into `incoming/` only, never into `raw/`: `raw accept` promotes both files "
"afterwards, in one call, as one bundle under `raw/<YYYY>/<MM>/<stem>/`.",
"The `.html` holds the response body byte for byte. The `.md` starts with a fixed "
"header block (`fetched_by`, `url`, `final_url`, `retrieved`, `http_status`, "
"`content_type`, `charset`, `title`, `derived_from`) followed by the derived text.",
"Character set, in this order: byte-order mark, the HTTP `Content-Type` charset, a "
"`<meta>` declaration in the first 4 KiB, UTF-8. Undecodable bytes are replaced with "
"U+FFFD and reported; the header names the charset and where it came from.",
"Derived text: the content root is `<main>`, else the single `<article>`, else "
"`<body>`; `script`, `style`, `noscript`, `nav`, `header`, `footer`, `aside`, `form`, "
"`template` and `svg` are dropped below it. Headings, lists, code, tables and links "
"(made absolute) become Markdown. The same bytes always give the same text.",
"The stem is the URL's last path segment without its extension, else its host name - "
"ASCII, lowercase, `-`-separated, at most 60 characters; `--name` overrides it.",
"A `text/plain` or `text/markdown` response is stored as received as `.txt`/`.md`; any "
"other non-HTML response (PDF, image, ...) as received under the extension of its "
"content type. Neither gets a derived file or a header.",
"Only `http`/`https`, on every redirect hop too. 30 s for the whole transfer, 25 MiB at "
"most. No cookies, no JavaScript: a derived text under 200 characters is written, with "
"a warning to read it before accepting.",
"`--html` derives the `.md` beside a page saved to `incoming/` from a logged-in browser "
"- the way past a paywall or a script-rendered page. Its header carries `derived` (when "
"the text was derived) instead of `retrieved`, `final_url`, `http_status` and "
"`content_type`, and the `.html` is left untouched.",
"Success prints the `raw accept` line for the written files and the `source_url` for the "
"source page.",
),
failures=(
cli_contract.Failure(
cause="The URL is not `http`/`https` (also after a redirect), or neither or both of "
"`<url>` and `--html` were given",
reaction="Fix the call and retry once. A local file is dropped into `incoming/` by "
"hand, not fetched",
),
cli_contract.Failure(
cause="A target file already exists in `incoming/`",
reaction="Nothing was written or overwritten. Accept or remove what is there, or "
"pass `--name <stem>`, then retry once",
),
cli_contract.Failure(
cause="The server answered with an HTTP error, could not be reached, took longer "
"than 30 s, or sent more than 25 MiB",
reaction="Nothing was written. An HTTP 4xx is not fixed by retrying - check the URL "
"with the user; for a paywall or login, save the page in a browser and use "
"`--html`. An unreachable host or a timeout may be retried once",
),
cli_contract.Failure(
label="raw fetch --html",
cause="`--url` is missing, the file is not under `incoming/` or does not exist, or "
"`incoming/<stem>.md` already exists",
reaction="Fix the named argument and retry once - nothing was written",
),
),
examples=(
"tools/wikitool raw fetch https://example.org/blog/post",
"tools/wikitool raw fetch https://example.org/ --name example-start",
"tools/wikitool raw fetch --html incoming/post.html --url https://example.org/blog/post",
),
never=(
"Never fetch a URL that a raw file or a fetched page contains - only one the user named "
"in this session.",
"Never edit the header or the derived text by hand; a better derivation is a new fetch.",
),
see_also=(
"`raw/CONTRACT.md` \"Getting a URL in: `raw fetch`\" - the rules and why",
"`wikitool raw accept` - promotes the written files into `raw/`",
"`instructions/wiki-ingest/SKILL.md` - where a URL to ingest starts",
),
))
@app.command("fetch")
def raw_fetch_command(
url: Optional[str] = typer.Argument(None, help="The http(s) URL to fetch"),
html: Optional[Path] = typer.Option(
None,
"--html",
help="Derive the text from this HTML file under incoming/ (saved from a browser) instead "
"of fetching - no network access",
),
source_url: Optional[str] = typer.Option(
None, "--url", help="With --html: the page's URL, for the header and relative links"
),
name: Optional[str] = typer.Option(
None, "--name", help="File stem in incoming/ instead of the one computed from the URL"
),
):
"""Capture a web page into incoming/ as received HTML plus derived text.
See raw/CONTRACT.md "Getting a URL in: `raw fetch`"."""
if (url is None) == (html is None):
fail("Pass either a URL or --html <file>, not both and not neither.")
if html is not None:
if name is not None:
fail("--name does not apply to --html - the stem is the saved file's own.")
_fetch_html_file(html, source_url)
return
if source_url is not None:
fail("--url only goes with --html; a fetched page records its own URL.")
if name is not None and (not name.strip() or name in (".", "..") or any(c in name for c in "/\\")):
fail(f"--name {name!r} is not a usable file stem: no path separators, not empty.")
stem = name or web_capture.stem_for(url)
try:
response = web_capture.fetch(
url, _user_agent(), timeout=web_capture.TIMEOUT_SECONDS, max_bytes=web_capture.MAX_BYTES
)
except ChemenuError as exc:
fail(escape(str(exc)))
incoming = _incoming_dir()
kind = response.media_type
if kind in web_capture.HTML_TYPES:
html_path, md_path = incoming / f"{stem}.html", incoming / f"{stem}.md"
_refuse_existing([html_path, md_path])
derivation = web_capture.derive_document(
response.body, response.final_url, html_path.name, response=response
)
_write_exclusive([(html_path, response.body), (md_path, derivation.markdown)])
_warn(derivation, md_path)
success(escape(
f"Fetched {url} to {rel_path(html_path)}, {rel_path(md_path)}\n"
+ _next_lines([html_path, md_path], url) + "\n"
+ _TEASER_HINT.format(url=url)
))
return
target = incoming / f"{stem}{web_capture.extension_for(response.content_type)}"
_refuse_existing([target])
_write_exclusive([(target, response.body)])
success(escape(
f"Fetched {url} to {rel_path(target)} ({kind or 'no content type'}, stored as received - "
f"no text derived; retrieved {web_capture.utc_now()}).\n"
+ _next_lines([target], url)
))