feat: raw fetch - a sanctioned intake for a URL into incoming/, HTML as received plus derived text (#120)
Files changed: - CHANGES.md - README.md - VERSION - instructions/wiki-ingest/SKILL.md - raw/CONTRACT.md - tools/CONTRACT.md - tools/README.md - tools/chemenu/cli_contract.py - tools/chemenu/commands/raw_cmd.py - tools/chemenu/tests/test_cli.py - tools/chemenu/tests/test_portability.py - tools/chemenu/tests/test_raw_fetch.py - tools/chemenu/web_capture.py Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01SnAJ7Z3CpVD3PRbN73QtU2
This commit is contained in:
1 parent
c0f324ff96
commit
fba263af68
13 files changed
+1571
-16
No files matched your search
@@ -70,15 +70,18 @@ from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
from rich.markup import escape
|
||||
|
||||
from chemenu import cli_contract, config
|
||||
from chemenu import cli_contract, config, web_capture
|
||||
from chemenu.commands._util import check_path_budget, fail, rel_path, success
|
||||
from chemenu.errors import ChemenuError
|
||||
from chemenu.frontmatter_io import write_page
|
||||
from chemenu.kb_scan import load_kb_pages
|
||||
from chemenu.provenance import citing_pages, source_pages_by_raw_file, source_raw_files
|
||||
from chemenu.type_resolver import resolver
|
||||
from chemenu.version import VersionError, read_version
|
||||
|
||||
app = typer.Typer(help="Promote raw material out of incoming/ into raw/.")
|
||||
app = typer.Typer(help="Bring raw material into incoming/, and promote it from there into raw/.")
|
||||
|
||||
# The type this command always promotes into eventually - hardcoded rather
|
||||
# than derived from --page, because there is no page yet in the common case
|
||||
@@ -118,8 +121,8 @@ def _validate_under_incoming(path: Path, incoming: Path) -> None:
|
||||
rel = path.relative_to(incoming)
|
||||
except ValueError:
|
||||
fail(
|
||||
f"{rel_path(path)} is not under incoming/ - `raw accept` only promotes files "
|
||||
"from there. See raw/CONTRACT.md."
|
||||
f"{rel_path(path)} is not under incoming/ - `raw accept` and `raw fetch --html` "
|
||||
"only take files from there. See raw/CONTRACT.md."
|
||||
)
|
||||
if len(rel.parts) < 1:
|
||||
fail(f"incoming/{rel.as_posix()} names no file.")
|
||||
@@ -658,3 +661,261 @@ def raw_accept_command(
|
||||
f" --set fidelity={fidelity} --set authority={authority} \\\n"
|
||||
" --set source_type=<category>"
|
||||
)
|
||||
|
||||
|
||||
# --- raw fetch ----------------------------------------------------------------
|
||||
#
|
||||
# The sanctioned intake for a URL the user names. It ends in `incoming/`, never
|
||||
# in `raw/`: promoting stays `raw accept`'s job, so `wiki-ingest` keeps its
|
||||
# order (the commitment question comes before the promotion) and the capture
|
||||
# fields are asked exactly once, there. The fetch and derivation themselves live
|
||||
# in `chemenu.web_capture`; this adapter decides file names and what the caller
|
||||
# is told.
|
||||
|
||||
_TEASER_HINT = (
|
||||
" If the text stops at a teaser (paywall, login, script-rendered page): save the page from a\n"
|
||||
" logged-in browser to incoming/ and run tools/wikitool raw fetch --html "
|
||||
"incoming/<file>.html --url {url}"
|
||||
)
|
||||
|
||||
|
||||
def _user_agent() -> str:
|
||||
try:
|
||||
version = str(read_version())
|
||||
except VersionError:
|
||||
version = "unknown"
|
||||
return f"chemenu-wikitool/{version} (raw fetch)"
|
||||
|
||||
|
||||
def _next_lines(files: list[Path], source_url: str) -> str:
|
||||
paths = " ".join(rel_path(f) for f in files)
|
||||
return (
|
||||
" Next (after the commitment question in wiki-ingest):\n"
|
||||
f" tools/wikitool raw accept --fidelity published --authority <value> {paths}\n"
|
||||
f" and on the source page: --set source_url={source_url}"
|
||||
)
|
||||
|
||||
|
||||
def _refuse_existing(targets: list[Path]) -> None:
|
||||
existing = [t for t in targets if t.exists()]
|
||||
if existing:
|
||||
fail(escape(
|
||||
f"{', '.join(rel_path(t) for t in existing)} already exists in incoming/ - raw fetch "
|
||||
"never overwrites. Accept or remove the file that is there, or pass --name <stem>."
|
||||
))
|
||||
|
||||
|
||||
def _write_exclusive(writes: list[tuple[Path, bytes]]) -> None:
|
||||
"""Create every file or none: `x` mode refuses a file that appeared since
|
||||
the check above, and a failure part-way removes what this call wrote."""
|
||||
written: list[Path] = []
|
||||
try:
|
||||
for path, data in writes:
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with path.open("xb") as handle:
|
||||
written.append(path)
|
||||
handle.write(data)
|
||||
except OSError as exc:
|
||||
for path in written:
|
||||
path.unlink(missing_ok=True)
|
||||
fail(escape(f"Could not write {rel_path(path)}: {exc}. Nothing was left in incoming/."))
|
||||
|
||||
|
||||
def _warn(derivation: web_capture.Derivation, md_path: Path) -> None:
|
||||
if derivation.decoded.replaced:
|
||||
typer.echo(
|
||||
f"WARN {derivation.decoded.replaced} undecodable byte(s) under "
|
||||
f"{derivation.decoded.charset} were replaced with U+FFFD in {rel_path(md_path)} - "
|
||||
"the .html beside it keeps the original bytes."
|
||||
)
|
||||
if derivation.text_chars < web_capture.SHORT_TEXT_CHARS:
|
||||
typer.echo(
|
||||
f"WARN The derived text is only {derivation.text_chars} characters - likely a "
|
||||
"script-rendered page, a login wall or an empty response. Read "
|
||||
f"{rel_path(md_path)} before accepting it."
|
||||
)
|
||||
|
||||
|
||||
def _fetch_html_file(html: Path, source_url: Optional[str]) -> None:
|
||||
if source_url is None:
|
||||
fail("--html needs --url <url>: it goes into the header, resolves relative links, and "
|
||||
"becomes source_url on the source page.")
|
||||
try:
|
||||
web_capture.check_url(source_url)
|
||||
except ChemenuError as exc:
|
||||
fail(escape(str(exc)))
|
||||
path = _resolve(html)
|
||||
_validate_under_incoming(path, _incoming_dir())
|
||||
if not path.is_file():
|
||||
fail(f"{rel_path(path)} does not exist or is not a file.")
|
||||
md_path = path.with_suffix(".md")
|
||||
if md_path == path:
|
||||
fail(f"{rel_path(path)} is already a .md file - --html takes the saved HTML page.")
|
||||
_refuse_existing([md_path])
|
||||
|
||||
derivation = web_capture.derive_document(path.read_bytes(), source_url, path.name, url=source_url)
|
||||
_write_exclusive([(md_path, derivation.markdown)])
|
||||
_warn(derivation, md_path)
|
||||
success(escape(
|
||||
f"Derived {rel_path(md_path)} from {rel_path(path)} (no network access).\n"
|
||||
+ _next_lines([path, md_path], source_url)
|
||||
))
|
||||
|
||||
|
||||
@cli_contract.record(cli_contract.CommandRecord(
|
||||
path="raw fetch",
|
||||
summary="Capture a web page the user names into `incoming/`: the HTML as received plus a "
|
||||
"derived text, for `raw accept` to promote.",
|
||||
synopsis=(
|
||||
cli_contract.Variant(
|
||||
usage="raw fetch <url> [--name <stem>]",
|
||||
notes="Fetch the page and write `incoming/<stem>.html` and `incoming/<stem>.md`",
|
||||
),
|
||||
cli_contract.Variant(
|
||||
usage="raw fetch --html incoming/<file>.html --url <url>",
|
||||
notes="Derive the `.md` from a page a human saved from their browser - no network "
|
||||
"access",
|
||||
),
|
||||
),
|
||||
properties=cli_contract.Properties(
|
||||
effect=cli_contract.Effect.WRITE,
|
||||
idempotent=cli_contract.Idempotent.NO,
|
||||
atomic="Yes for what it leaves behind - one or two new files in `incoming/`, created "
|
||||
"exclusively; a failure on the second removes the first",
|
||||
budget=cli_contract.Budget.COUNTED,
|
||||
network=cli_contract.Network.YES,
|
||||
),
|
||||
notes=(
|
||||
"Writes into `incoming/` only, never into `raw/`: `raw accept` promotes both files "
|
||||
"afterwards, in one call, as one bundle under `raw/<YYYY>/<MM>/<stem>/`.",
|
||||
"The `.html` holds the response body byte for byte. The `.md` starts with a fixed "
|
||||
"header block (`fetched_by`, `url`, `final_url`, `retrieved`, `http_status`, "
|
||||
"`content_type`, `charset`, `title`, `derived_from`) followed by the derived text.",
|
||||
"Character set, in this order: byte-order mark, the HTTP `Content-Type` charset, a "
|
||||
"`<meta>` declaration in the first 4 KiB, UTF-8. Undecodable bytes are replaced with "
|
||||
"U+FFFD and reported; the header names the charset and where it came from.",
|
||||
"Derived text: the content root is `<main>`, else the single `<article>`, else "
|
||||
"`<body>`; `script`, `style`, `noscript`, `nav`, `header`, `footer`, `aside`, `form`, "
|
||||
"`template` and `svg` are dropped below it. Headings, lists, code, tables and links "
|
||||
"(made absolute) become Markdown. The same bytes always give the same text.",
|
||||
"The stem is the URL's last path segment without its extension, else its host name - "
|
||||
"ASCII, lowercase, `-`-separated, at most 60 characters; `--name` overrides it.",
|
||||
"A `text/plain` or `text/markdown` response is stored as received as `.txt`/`.md`; any "
|
||||
"other non-HTML response (PDF, image, ...) as received under the extension of its "
|
||||
"content type. Neither gets a derived file or a header.",
|
||||
"Only `http`/`https`, on every redirect hop too. 30 s for the whole transfer, 25 MiB at "
|
||||
"most. No cookies, no JavaScript: a derived text under 200 characters is written, with "
|
||||
"a warning to read it before accepting.",
|
||||
"`--html` derives the `.md` beside a page saved to `incoming/` from a logged-in browser "
|
||||
"- the way past a paywall or a script-rendered page. Its header carries `derived` (when "
|
||||
"the text was derived) instead of `retrieved`, `final_url`, `http_status` and "
|
||||
"`content_type`, and the `.html` is left untouched.",
|
||||
"Success prints the `raw accept` line for the written files and the `source_url` for the "
|
||||
"source page.",
|
||||
),
|
||||
failures=(
|
||||
cli_contract.Failure(
|
||||
cause="The URL is not `http`/`https` (also after a redirect), or neither or both of "
|
||||
"`<url>` and `--html` were given",
|
||||
reaction="Fix the call and retry once. A local file is dropped into `incoming/` by "
|
||||
"hand, not fetched",
|
||||
),
|
||||
cli_contract.Failure(
|
||||
cause="A target file already exists in `incoming/`",
|
||||
reaction="Nothing was written or overwritten. Accept or remove what is there, or "
|
||||
"pass `--name <stem>`, then retry once",
|
||||
),
|
||||
cli_contract.Failure(
|
||||
cause="The server answered with an HTTP error, could not be reached, took longer "
|
||||
"than 30 s, or sent more than 25 MiB",
|
||||
reaction="Nothing was written. An HTTP 4xx is not fixed by retrying - check the URL "
|
||||
"with the user; for a paywall or login, save the page in a browser and use "
|
||||
"`--html`. An unreachable host or a timeout may be retried once",
|
||||
),
|
||||
cli_contract.Failure(
|
||||
label="raw fetch --html",
|
||||
cause="`--url` is missing, the file is not under `incoming/` or does not exist, or "
|
||||
"`incoming/<stem>.md` already exists",
|
||||
reaction="Fix the named argument and retry once - nothing was written",
|
||||
),
|
||||
),
|
||||
examples=(
|
||||
"tools/wikitool raw fetch https://example.org/blog/post",
|
||||
"tools/wikitool raw fetch https://example.org/ --name example-start",
|
||||
"tools/wikitool raw fetch --html incoming/post.html --url https://example.org/blog/post",
|
||||
),
|
||||
never=(
|
||||
"Never fetch a URL that a raw file or a fetched page contains - only one the user named "
|
||||
"in this session.",
|
||||
"Never edit the header or the derived text by hand; a better derivation is a new fetch.",
|
||||
),
|
||||
see_also=(
|
||||
"`raw/CONTRACT.md` \"Getting a URL in: `raw fetch`\" - the rules and why",
|
||||
"`wikitool raw accept` - promotes the written files into `raw/`",
|
||||
"`instructions/wiki-ingest/SKILL.md` - where a URL to ingest starts",
|
||||
),
|
||||
))
|
||||
@app.command("fetch")
|
||||
def raw_fetch_command(
|
||||
url: Optional[str] = typer.Argument(None, help="The http(s) URL to fetch"),
|
||||
html: Optional[Path] = typer.Option(
|
||||
None,
|
||||
"--html",
|
||||
help="Derive the text from this HTML file under incoming/ (saved from a browser) instead "
|
||||
"of fetching - no network access",
|
||||
),
|
||||
source_url: Optional[str] = typer.Option(
|
||||
None, "--url", help="With --html: the page's URL, for the header and relative links"
|
||||
),
|
||||
name: Optional[str] = typer.Option(
|
||||
None, "--name", help="File stem in incoming/ instead of the one computed from the URL"
|
||||
),
|
||||
):
|
||||
"""Capture a web page into incoming/ as received HTML plus derived text.
|
||||
See raw/CONTRACT.md "Getting a URL in: `raw fetch`"."""
|
||||
if (url is None) == (html is None):
|
||||
fail("Pass either a URL or --html <file>, not both and not neither.")
|
||||
if html is not None:
|
||||
if name is not None:
|
||||
fail("--name does not apply to --html - the stem is the saved file's own.")
|
||||
_fetch_html_file(html, source_url)
|
||||
return
|
||||
if source_url is not None:
|
||||
fail("--url only goes with --html; a fetched page records its own URL.")
|
||||
|
||||
if name is not None and (not name.strip() or name in (".", "..") or any(c in name for c in "/\\")):
|
||||
fail(f"--name {name!r} is not a usable file stem: no path separators, not empty.")
|
||||
stem = name or web_capture.stem_for(url)
|
||||
|
||||
try:
|
||||
response = web_capture.fetch(
|
||||
url, _user_agent(), timeout=web_capture.TIMEOUT_SECONDS, max_bytes=web_capture.MAX_BYTES
|
||||
)
|
||||
except ChemenuError as exc:
|
||||
fail(escape(str(exc)))
|
||||
|
||||
incoming = _incoming_dir()
|
||||
kind = response.media_type
|
||||
if kind in web_capture.HTML_TYPES:
|
||||
html_path, md_path = incoming / f"{stem}.html", incoming / f"{stem}.md"
|
||||
_refuse_existing([html_path, md_path])
|
||||
derivation = web_capture.derive_document(
|
||||
response.body, response.final_url, html_path.name, response=response
|
||||
)
|
||||
_write_exclusive([(html_path, response.body), (md_path, derivation.markdown)])
|
||||
_warn(derivation, md_path)
|
||||
success(escape(
|
||||
f"Fetched {url} to {rel_path(html_path)}, {rel_path(md_path)}\n"
|
||||
+ _next_lines([html_path, md_path], url) + "\n"
|
||||
+ _TEASER_HINT.format(url=url)
|
||||
))
|
||||
return
|
||||
|
||||
target = incoming / f"{stem}{web_capture.extension_for(response.content_type)}"
|
||||
_refuse_existing([target])
|
||||
_write_exclusive([(target, response.body)])
|
||||
success(escape(
|
||||
f"Fetched {url} to {rel_path(target)} ({kind or 'no content type'}, stored as received - "
|
||||
f"no text derived; retrieved {web_capture.utc_now()}).\n"
|
||||
+ _next_lines([target], url)
|
||||
))
|
||||
Reference in new issue
Block a user