import uuid from urllib.parse import urljoin import httpx import rfc3987 from bs4 import BeautifulSoup from flask import current_app from .. import APP_NAME, VERSION from .. import DATABASE as db from .. import HUEY as huey from ..models import ReceivedWebmention from ..url_security import AddressResolutionError, is_public_url class VerificationError(Exception): """The source permanently failed ReceivedWebmention verification.""" class SourceGoneError(VerificationError): """The source explicitly reports that it has been removed.""" class TemporaryFetchError(Exception): """Fetching the source may succeed when retried later.""" IRI_PATTERN = rfc3987.get_compiled_pattern("IRI") HTML_URL_ATTRIBUTES = { "href": {"a", "area", "link"}, "src": { "audio", "embed", "iframe", "img", 'input[type="image" i]', "script", "audio source", "video source", "track", "video", }, "cite": {"blockquote", "del", "ins", "q"}, } def html_mentions_target(body: bytes, source_url: str, target_url: str) -> bool: """ Check whether an HTML document mentions the target URL. URL references are resolved against the document's ``base`` element when present, or against the source URL otherwise. :param body: HTML document body. :param source_url: URL from which the document was retrieved. :param target_url: Exact URL to look for. :return: Whether a supported HTML URL attribute references the target URL. """ document = BeautifulSoup(body, "html.parser") base_url = source_url if (base_element := document.select_one("base[href]")) is not None and isinstance( base_href := base_element.get("href"), str ): base_url = urljoin(source_url, base_href.strip()) for attribute, selectors in HTML_URL_ATTRIBUTES.items(): selector = ", ".join( [f"{selector_string}[{attribute}]" for selector_string in selectors] ) for element in document.select(selector): if not isinstance(reference := element.get(attribute), str): continue if urljoin(base_url, reference.strip()) == target_url: return True return False def text_mentions_target(body: str, target_url: str) -> bool: """ Check whether plain text contains the exact target IRI. :param body: Plain-text document body. :param target_url: Exact IRI to look for. :return: Whether the target IRI occurs in the document. """ return any(match.group() == target_url for match in IRI_PATTERN.finditer(body)) def ensure_public_request(request: httpx.Request) -> None: """ Ensure that an HTTP request targets a public network address. :param request: HTTP request to validate. :raises VerificationError: If the URL has no hostname or resolves to a non-public address. :raises TemporaryFetchError: If the hostname cannot be resolved. """ try: if not is_public_url(str(request.url)): raise VerificationError("Source resolves to a non-public address") except ValueError as exc: raise VerificationError("Source URL has no hostname") from exc except AddressResolutionError as exc: raise TemporaryFetchError("Source hostname could not be resolved") from exc def fetch_source(source_url: str) -> tuple[httpx.Response, bytes]: """ Fetch a webmention source document. The request follows redirects while enforcing the configured timeout, redirect limit, and maximum response size. :param source_url: URL of the source document. :return: HTTP response and response body. :raises SourceGoneError: If the source returns HTTP 410. :raises TemporaryFetchError: If the source returns a temporary HTTP error. :raises VerificationError: If the source returns another unsuccessful HTTP status or exceeds the configured maximum response size. """ with ( httpx.Client( headers={ "Accept": "text/html, application/xhtml+xml;q=0.9, text/plain;q=0.8", "User-Agent": f"{APP_NAME}/{VERSION} ReceivedWebmention", }, timeout=httpx.Timeout( current_app.config.get("WEBMENTIONS_SSG_REQUEST_TIMEOUT", 5.0) ), follow_redirects=True, max_redirects=current_app.config.get("WEBMENTIONS_SSG_MAX_REDIRECTS", 20), trust_env=False, event_hooks={"request": [ensure_public_request]}, ) as client, client.stream("GET", source_url) as response, ): match response.status_code: case 200: pass case 410: raise SourceGoneError("Source returned HTTP 410") case status: if status in {408, 425, 429} or 500 <= status <= 599: raise TemporaryFetchError(f"Source returned HTTP {status}") else: raise VerificationError(f"Source returned HTTP {status}") max_source_bytes = current_app.config.get( "WEBMENTIONS_SSG_MAX_SOURCE_BYTES", 1_000_000 ) if (content_length := response.headers.get("Content-Length")) is not None: try: if int(content_length) > max_source_bytes: raise VerificationError("Source document is too large") except ValueError: pass body = bytearray() for chunk in response.iter_bytes(chunk_size=64 * 1024): body.extend(chunk) if len(body) > max_source_bytes: raise VerificationError("Source document is too large") return response, bytes(body) def source_mentions_target(source_url: str, target_url: str) -> bool: """ Fetch a source and check whether it mentions the target URL. HTML and XHTML sources are inspected for supported URL attributes, while plain-text sources are searched for the exact target IRI. :param source_url: URL of the source document. :param target_url: Exact target URL to look for. :return: Whether the source mentions the target URL. :raises VerificationError: If the source has an unsupported or missing content type. :raises SourceGoneError: If the source explicitly reports that it is gone. :raises TemporaryFetchError: If fetching the source fails temporarily. """ response, body = fetch_source(source_url) media_type = ( response.headers.get("Content-Type", "").partition(";")[0].strip().lower() ) match media_type: case "text/html" | "application/xhtml+xml": return html_mentions_target(body, str(response.url), target_url) case "text/plain": try: decoded_body = body.decode( response.encoding or "utf-8", errors="replace" ) except LookupError: decoded_body = body.decode("utf-8", errors="replace") return text_mentions_target(decoded_body, target_url) case _: raise VerificationError( f"Unsupported source content type: {media_type or 'missing'}" ) @huey.task(retries=2, retry_delay=50) def verify_webmention(webmention_uuid: uuid.UUID) -> None: """ Verify a received webmention and store its verification status. Temporary fetch failures are stored before being re-raised so that Huey can retry the task. :param webmention_uuid: Identifier of the received webmention to verify. :raises TemporaryFetchError: If fetching the source fails temporarily. :raises httpx.RequestError: If an HTTP request error occurs while fetching the source. """ webmention = db.session.get(ReceivedWebmention, webmention_uuid) if webmention is None: current_app.logger.warning( "Cannot verify unknown ReceivedWebmention %s", webmention_uuid ) return webmention.status = "verifying" webmention.failure_reason = None db.session.commit() try: mentions_target = source_mentions_target(webmention.source, webmention.target) except SourceGoneError as exc: webmention.status = "deleted" webmention.failure_reason = str(exc) except VerificationError as exc: webmention.status = "failed" webmention.failure_reason = str(exc) except (TemporaryFetchError, httpx.RequestError) as exc: webmention.status = "failed" webmention.failure_reason = str(exc) or "Source could not be fetched" db.session.commit() # Huey retries the task because the exception escapes. raise else: if mentions_target: webmention.status = "verified" webmention.failure_reason = None else: webmention.status = "deleted" webmention.failure_reason = "Source does not mention target" db.session.commit()