aboutsummaryrefslogtreecommitdiff
path: root/webmentions_ssg/tasks/receiver.py
diff options
context:
space:
mode:
Diffstat (limited to 'webmentions_ssg/tasks/receiver.py')
-rw-r--r--webmentions_ssg/tasks/receiver.py155
1 files changed, 105 insertions, 50 deletions
diff --git a/webmentions_ssg/tasks/receiver.py b/webmentions_ssg/tasks/receiver.py
index 39def73..4aba3d8 100644
--- a/webmentions_ssg/tasks/receiver.py
+++ b/webmentions_ssg/tasks/receiver.py
@@ -47,7 +47,17 @@ HTML_URL_ATTRIBUTES = {
def html_mentions_target(body: bytes, source_url: str, target_url: str) -> bool:
- """Check valid HTML URL attributes for the exact target URL."""
+ """
+ Check whether an HTML document mentions the target URL.
+
+ URL references are resolved against the document's ``base`` element when
+ present, or against the source URL otherwise.
+
+ :param body: HTML document body.
+ :param source_url: URL from which the document was retrieved.
+ :param target_url: Exact URL to look for.
+ :return: Whether a supported HTML URL attribute references the target URL.
+ """
document = BeautifulSoup(body, "html.parser")
base_url = source_url
@@ -59,12 +69,7 @@ def html_mentions_target(body: bytes, source_url: str, target_url: str) -> bool:
for attribute, selectors in HTML_URL_ATTRIBUTES.items():
selector = ", ".join(
- [
- "{selector}[{attribute}]".format(
- selector=selector_string, attribute=attribute
- )
- for selector_string in selectors
- ]
+ [f"{selector_string}[{attribute}]" for selector_string in selectors]
)
for element in document.select(selector):
@@ -78,12 +83,25 @@ def html_mentions_target(body: bytes, source_url: str, target_url: str) -> bool:
def text_mentions_target(body: str, target_url: str) -> bool:
- """Check whether plain text contains the exact target IRI."""
+ """
+ Check whether plain text contains the exact target IRI.
+
+ :param body: Plain-text document body.
+ :param target_url: Exact IRI to look for.
+ :return: Whether the target IRI occurs in the document.
+ """
return any(match.group() == target_url for match in IRI_PATTERN.finditer(body))
def ensure_public_request(request: httpx.Request) -> None:
- """Prevent requests to non-public network addresses."""
+ """
+ Ensure that an HTTP request targets a public network address.
+
+ :param request: HTTP request to validate.
+ :raises VerificationError: If the URL has no hostname or resolves to a
+ non-public address.
+ :raises TemporaryFetchError: If the hostname cannot be resolved.
+ """
try:
if not is_public_url(str(request.url)):
raise VerificationError("Source resolves to a non-public address")
@@ -94,56 +112,83 @@ def ensure_public_request(request: httpx.Request) -> None:
def fetch_source(source_url: str) -> tuple[httpx.Response, bytes]:
- """Fetch a source with limits on redirects, time, and response size."""
+ """
+ Fetch a webmention source document.
- with httpx.Client(
- headers={
- "Accept": "text/html, application/xhtml+xml;q=0.9, text/plain;q=0.8",
- "User-Agent": f"{APP_NAME}/{VERSION} ReceivedWebmention",
- },
- timeout=httpx.Timeout(
- current_app.config.get("WEBMENTIONS_SSG_REQUEST_TIMEOUT", 5.0)
- ),
- follow_redirects=True,
- max_redirects=current_app.config.get("WEBMENTIONS_SSG_MAX_REDIRECTS", 20),
- trust_env=False,
- event_hooks={"request": [ensure_public_request]},
- ) as client:
- with client.stream("GET", source_url) as response:
- match response.status_code:
- case 200:
- pass
- case 410:
- raise SourceGoneError("Source returned HTTP 410")
- case status:
- if status in {408, 425, 429} or 500 <= status <= 599:
- raise TemporaryFetchError(f"Source returned HTTP {status}")
- else:
- raise VerificationError(f"Source returned HTTP {status}")
+ The request follows redirects while enforcing the configured timeout,
+ redirect limit, and maximum response size.
- max_source_bytes = current_app.config.get(
- "WEBMENTIONS_SSG_MAX_SOURCE_BYTES", 1_000_000
- )
+ :param source_url: URL of the source document.
+ :return: HTTP response and response body.
+ :raises SourceGoneError: If the source returns HTTP 410.
+ :raises TemporaryFetchError: If the source returns a temporary HTTP error.
+ :raises VerificationError: If the source returns another unsuccessful HTTP
+ status or exceeds the configured maximum response size.
+ """
- if (content_length := response.headers.get("Content-Length")) is not None:
- try:
- if int(content_length) > max_source_bytes:
- raise VerificationError("Source document is too large")
- except ValueError:
- pass
+ with (
+ httpx.Client(
+ headers={
+ "Accept": "text/html, application/xhtml+xml;q=0.9, text/plain;q=0.8",
+ "User-Agent": f"{APP_NAME}/{VERSION} ReceivedWebmention",
+ },
+ timeout=httpx.Timeout(
+ current_app.config.get("WEBMENTIONS_SSG_REQUEST_TIMEOUT", 5.0)
+ ),
+ follow_redirects=True,
+ max_redirects=current_app.config.get("WEBMENTIONS_SSG_MAX_REDIRECTS", 20),
+ trust_env=False,
+ event_hooks={"request": [ensure_public_request]},
+ ) as client,
+ client.stream("GET", source_url) as response,
+ ):
+ match response.status_code:
+ case 200:
+ pass
+ case 410:
+ raise SourceGoneError("Source returned HTTP 410")
+ case status:
+ if status in {408, 425, 429} or 500 <= status <= 599:
+ raise TemporaryFetchError(f"Source returned HTTP {status}")
+ else:
+ raise VerificationError(f"Source returned HTTP {status}")
- body = bytearray()
- for chunk in response.iter_bytes(chunk_size=64 * 1024):
- body.extend(chunk)
+ max_source_bytes = current_app.config.get(
+ "WEBMENTIONS_SSG_MAX_SOURCE_BYTES", 1_000_000
+ )
- if len(body) > max_source_bytes:
+ if (content_length := response.headers.get("Content-Length")) is not None:
+ try:
+ if int(content_length) > max_source_bytes:
raise VerificationError("Source document is too large")
+ except ValueError:
+ pass
+
+ body = bytearray()
+ for chunk in response.iter_bytes(chunk_size=64 * 1024):
+ body.extend(chunk)
- return response, bytes(body)
+ if len(body) > max_source_bytes:
+ raise VerificationError("Source document is too large")
+
+ return response, bytes(body)
def source_mentions_target(source_url: str, target_url: str) -> bool:
- """Fetch the source and verify it according to its media type."""
+ """
+ Fetch a source and check whether it mentions the target URL.
+
+ HTML and XHTML sources are inspected for supported URL attributes, while
+ plain-text sources are searched for the exact target IRI.
+
+ :param source_url: URL of the source document.
+ :param target_url: Exact target URL to look for.
+ :return: Whether the source mentions the target URL.
+ :raises VerificationError: If the source has an unsupported or missing
+ content type.
+ :raises SourceGoneError: If the source explicitly reports that it is gone.
+ :raises TemporaryFetchError: If fetching the source fails temporarily.
+ """
response, body = fetch_source(source_url)
@@ -170,7 +215,17 @@ def source_mentions_target(source_url: str, target_url: str) -> bool:
@huey.task(retries=2, retry_delay=50)
def verify_webmention(webmention_uuid: uuid.UUID) -> None:
- """Verify a ReceivedWebmention and store the result."""
+ """
+ Verify a received webmention and store its verification status.
+
+ Temporary fetch failures are stored before being re-raised so that Huey can
+ retry the task.
+
+ :param webmention_uuid: Identifier of the received webmention to verify.
+ :raises TemporaryFetchError: If fetching the source fails temporarily.
+ :raises httpx.RequestError: If an HTTP request error occurs while fetching
+ the source.
+ """
webmention = db.session.get(ReceivedWebmention, webmention_uuid)