From 67eff0a854da010cb6ecd119d84238ec3e119272 Mon Sep 17 00:00:00 2001 From: Dennis Fink Date: Sun, 9 Aug 2026 14:15:29 +0200 Subject: Implement received Webmention handling Add the Flask application setup, database models and migrations, authentication, and configuration for development and testing. Implement asynchronous Webmention verification with Huey, including HTML and plain-text source validation, retries, status tracking, and size limits. Add status, login, and paginated received-Webmention views together with comprehensive tests for forms, views, and receiver tasks. --- webmentions_ssg/tasks/receiver.py | 225 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 225 insertions(+) create mode 100644 webmentions_ssg/tasks/receiver.py (limited to 'webmentions_ssg/tasks/receiver.py') diff --git a/webmentions_ssg/tasks/receiver.py b/webmentions_ssg/tasks/receiver.py new file mode 100644 index 0000000..ea48299 --- /dev/null +++ b/webmentions_ssg/tasks/receiver.py @@ -0,0 +1,225 @@ +import uuid +from urllib.parse import urljoin + +import httpx +import rfc3987 +from bs4 import BeautifulSoup +from flask import current_app + +from .. import APP_NAME, VERSION +from .. import DATABASE as db +from .. import HUEY as huey +from ..models import ReceivedWebmention + + +class VerificationError(Exception): + """The source permanently failed ReceivedWebmention verification.""" + + +class SourceGoneError(VerificationError): + """The source explicitly reports that it has been removed.""" + + +class TemporaryFetchError(Exception): + """Fetching the source may succeed when retried later.""" + + +IRI_PATTERN = rfc3987.get_compiled_pattern("IRI") + + +HTML_URL_ATTRIBUTES = { + "href": {"a", "area", "link"}, + "src": { + "audio", + "embed", + "iframe", + "img", + 'input[type="image" i]', + "script", + "audio source", + "video source", + "track", + "video", + }, + "cite": { + "blockquote", + "del", + "ins", + "q", + }, +} + + +def html_mentions_target( + body: bytes, + source_url: str, + target_url: str, +) -> bool: + """Check valid HTML URL attributes for the exact target URL.""" + + document = BeautifulSoup(body, "html.parser") + base_url = source_url + + if (base_element := document.select_one("base[href]")) is not None and isinstance( + base_href := base_element.get("href"), + str, + ): + base_url = urljoin( + source_url, + base_href.strip(), + ) + + for attribute, selectors in HTML_URL_ATTRIBUTES.items(): + selector = ", ".join( + [ + "{selector}[{attribute}]".format( + selector=selector_string, attribute=attribute + ) + for selector_string in selectors + ] + ) + + for element in document.select(selector): + if not isinstance( + reference := element.get(attribute), + str, + ): + continue + + if ( + urljoin( + base_url, + reference.strip(), + ) + == target_url + ): + return True + + return False + + +def text_mentions_target(body: str, target_url: str) -> bool: + """Check whether plain text contains the exact target IRI.""" + return any(match.group() == target_url for match in IRI_PATTERN.finditer(body)) + + +def fetch_source(source_url: str) -> tuple[httpx.Response, bytes]: + """Fetch a source with limits on redirects, time, and response size.""" + + with httpx.Client( + headers={ + "Accept": "text/html, application/xhtml+xml;q=0.9, text/plain;q=0.8", + "User-Agent": f"{APP_NAME}/{VERSION} ReceivedWebmention", + }, + timeout=httpx.Timeout( + current_app.config.get("WEBMENTIONS_SSG_REQUEST_TIMEOUT", 5.0) + ), + follow_redirects=True, + max_redirects=current_app.config.get("WEBMENTIONS_SSG_MAX_REDIRECTS", 20), + trust_env=False, + ) as client: + with client.stream("GET", source_url) as response: + match response.status_code: + case 200: + pass + case 410: + raise SourceGoneError("Source returned HTTP 410") + case status: + if status in {408, 425, 429} or 500 <= status <= 599: + raise TemporaryFetchError(f"Source returned HTTP {status}") + else: + raise VerificationError(f"Source returned HTTP {status}") + + max_source_bytes = current_app.config.get( + "WEBMENTIONS_SSG_MAX_SOURCE_BYTES", + 1_000_000, + ) + + if (content_length := response.headers.get("Content-Length")) is not None: + try: + if int(content_length) > max_source_bytes: + raise VerificationError("Source document is too large") + except ValueError: + pass + + body = bytearray() + for chunk in response.iter_bytes(chunk_size=64 * 1024): + body.extend(chunk) + + if len(body) > max_source_bytes: + raise VerificationError("Source document is too large") + + return response, bytes(body) + + +def source_mentions_target(source_url: str, target_url: str) -> bool: + """Fetch the source and verify it according to its media type.""" + + response, body = fetch_source(source_url) + + media_type = ( + response.headers.get("Content-Type", "").partition(";")[0].strip().lower() + ) + + match media_type: + case "text/html" | "application/xhtml+xml": + return html_mentions_target(body, str(response.url), target_url) + case "text/plain": + try: + decoded_body = body.decode( + response.encoding or "utf-8", + errors="replace", + ) + except LookupError: + decoded_body = body.decode( + "utf-8", + errors="replace", + ) + return text_mentions_target(decoded_body, target_url) + case _: + raise VerificationError( + f"Unsupported source content type: {media_type or 'missing'}" + ) + + +@huey.task(retries=2, retry_delay=50) +def verify_webmention(webmention_uuid: uuid.UUID) -> None: + """Verify a ReceivedWebmention and store the result.""" + + webmention = db.session.get(ReceivedWebmention, webmention_uuid) + + if webmention is None: + current_app.logger.warning( + "Cannot verify unknown ReceivedWebmention %s", + webmention_uuid, + ) + return + + webmention.status = "verifying" + webmention.failure_reason = None + db.session.commit() + + try: + mentions_target = source_mentions_target(webmention.source, webmention.target) + except SourceGoneError as exc: + webmention.status = "deleted" + webmention.failure_reason = str(exc) + except VerificationError as exc: + webmention.status = "failed" + webmention.failure_reason = str(exc) + except (TemporaryFetchError, httpx.RequestError) as exc: + webmention.status = "failed" + webmention.failure_reason = str(exc) or "Source could not be fetched" + db.session.commit() + + # Huey retries the task because the exception escapes. + raise + else: + if mentions_target: + webmention.status = "verified" + webmention.failure_reason = None + else: + webmention.status = "deleted" + webmention.failure_reason = "Source does not mention target" + + db.session.commit() -- cgit v1.3.1