import uuid from urllib.parse import urljoin import httpx import rfc3987 from bs4 import BeautifulSoup from flask import current_app from .. import APP_NAME, VERSION from .. import DATABASE as db from .. import HUEY as huey from ..models import ReceivedWebmention class VerificationError(Exception): """The source permanently failed ReceivedWebmention verification.""" class SourceGoneError(VerificationError): """The source explicitly reports that it has been removed.""" class TemporaryFetchError(Exception): """Fetching the source may succeed when retried later.""" IRI_PATTERN = rfc3987.get_compiled_pattern("IRI") HTML_URL_ATTRIBUTES = { "href": {"a", "area", "link"}, "src": { "audio", "embed", "iframe", "img", 'input[type="image" i]', "script", "audio source", "video source", "track", "video", }, "cite": { "blockquote", "del", "ins", "q", }, } def html_mentions_target( body: bytes, source_url: str, target_url: str, ) -> bool: """Check valid HTML URL attributes for the exact target URL.""" document = BeautifulSoup(body, "html.parser") base_url = source_url if (base_element := document.select_one("base[href]")) is not None and isinstance( base_href := base_element.get("href"), str, ): base_url = urljoin( source_url, base_href.strip(), ) for attribute, selectors in HTML_URL_ATTRIBUTES.items(): selector = ", ".join( [ "{selector}[{attribute}]".format( selector=selector_string, attribute=attribute ) for selector_string in selectors ] ) for element in document.select(selector): if not isinstance( reference := element.get(attribute), str, ): continue if ( urljoin( base_url, reference.strip(), ) == target_url ): return True return False def text_mentions_target(body: str, target_url: str) -> bool: """Check whether plain text contains the exact target IRI.""" return any(match.group() == target_url for match in IRI_PATTERN.finditer(body)) def fetch_source(source_url: str) -> tuple[httpx.Response, bytes]: """Fetch a source with limits on redirects, time, and response size.""" with httpx.Client( headers={ "Accept": "text/html, application/xhtml+xml;q=0.9, text/plain;q=0.8", "User-Agent": f"{APP_NAME}/{VERSION} ReceivedWebmention", }, timeout=httpx.Timeout( current_app.config.get("WEBMENTIONS_SSG_REQUEST_TIMEOUT", 5.0) ), follow_redirects=True, max_redirects=current_app.config.get("WEBMENTIONS_SSG_MAX_REDIRECTS", 20), trust_env=False, ) as client: with client.stream("GET", source_url) as response: match response.status_code: case 200: pass case 410: raise SourceGoneError("Source returned HTTP 410") case status: if status in {408, 425, 429} or 500 <= status <= 599: raise TemporaryFetchError(f"Source returned HTTP {status}") else: raise VerificationError(f"Source returned HTTP {status}") max_source_bytes = current_app.config.get( "WEBMENTIONS_SSG_MAX_SOURCE_BYTES", 1_000_000, ) if (content_length := response.headers.get("Content-Length")) is not None: try: if int(content_length) > max_source_bytes: raise VerificationError("Source document is too large") except ValueError: pass body = bytearray() for chunk in response.iter_bytes(chunk_size=64 * 1024): body.extend(chunk) if len(body) > max_source_bytes: raise VerificationError("Source document is too large") return response, bytes(body) def source_mentions_target(source_url: str, target_url: str) -> bool: """Fetch the source and verify it according to its media type.""" response, body = fetch_source(source_url) media_type = ( response.headers.get("Content-Type", "").partition(";")[0].strip().lower() ) match media_type: case "text/html" | "application/xhtml+xml": return html_mentions_target(body, str(response.url), target_url) case "text/plain": try: decoded_body = body.decode( response.encoding or "utf-8", errors="replace", ) except LookupError: decoded_body = body.decode( "utf-8", errors="replace", ) return text_mentions_target(decoded_body, target_url) case _: raise VerificationError( f"Unsupported source content type: {media_type or 'missing'}" ) @huey.task(retries=2, retry_delay=50) def verify_webmention(webmention_uuid: uuid.UUID) -> None: """Verify a ReceivedWebmention and store the result.""" webmention = db.session.get(ReceivedWebmention, webmention_uuid) if webmention is None: current_app.logger.warning( "Cannot verify unknown ReceivedWebmention %s", webmention_uuid, ) return webmention.status = "verifying" webmention.failure_reason = None db.session.commit() try: mentions_target = source_mentions_target(webmention.source, webmention.target) except SourceGoneError as exc: webmention.status = "deleted" webmention.failure_reason = str(exc) except VerificationError as exc: webmention.status = "failed" webmention.failure_reason = str(exc) except (TemporaryFetchError, httpx.RequestError) as exc: webmention.status = "failed" webmention.failure_reason = str(exc) or "Source could not be fetched" db.session.commit() # Huey retries the task because the exception escapes. raise else: if mentions_target: webmention.status = "verified" webmention.failure_reason = None else: webmention.status = "deleted" webmention.failure_reason = "Source does not mention target" db.session.commit()