1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
|
# SPDX-FileCopyrightText: 2026 Dennis Fink <me+coding@dennisfink.me>
#
# SPDX-License-Identifier: BSD-3-Clause
import uuid
from urllib.parse import urljoin
import httpx
import rfc3987
from bs4 import BeautifulSoup
from flask import current_app
from .. import APP_NAME, VERSION
from .. import DATABASE as db
from .. import HUEY as huey
from ..models import ReceivedWebmention
from ..url_security import AddressResolutionError, is_public_url
class VerificationError(Exception):
"""The source permanently failed ReceivedWebmention verification."""
class SourceGoneError(VerificationError):
"""The source explicitly reports that it has been removed."""
class TemporaryFetchError(Exception):
"""Fetching the source may succeed when retried later."""
IRI_PATTERN = rfc3987.get_compiled_pattern("IRI")
HTML_URL_ATTRIBUTES = {
"href": {"a", "area", "link"},
"src": {
"audio",
"embed",
"iframe",
"img",
'input[type="image" i]',
"script",
"audio source",
"video source",
"track",
"video",
},
"cite": {"blockquote", "del", "ins", "q"},
}
def html_mentions_target(body: bytes, source_url: str, target_url: str) -> bool:
"""
Check whether an HTML document mentions the target URL.
URL references are resolved against the document's ``base`` element when
present, or against the source URL otherwise.
:param body: HTML document body.
:param source_url: URL from which the document was retrieved.
:param target_url: Exact URL to look for.
:return: Whether a supported HTML URL attribute references the target URL.
"""
document = BeautifulSoup(body, "html.parser")
base_url = source_url
if (base_element := document.select_one("base[href]")) is not None and isinstance(
base_href := base_element.get("href"), str
):
base_url = urljoin(source_url, base_href.strip())
for attribute, selectors in HTML_URL_ATTRIBUTES.items():
selector = ", ".join(
[f"{selector_string}[{attribute}]" for selector_string in selectors]
)
for element in document.select(selector):
if not isinstance(reference := element.get(attribute), str):
continue
if urljoin(base_url, reference.strip()) == target_url:
return True
return False
def text_mentions_target(body: str, target_url: str) -> bool:
"""
Check whether plain text contains the exact target IRI.
:param body: Plain-text document body.
:param target_url: Exact IRI to look for.
:return: Whether the target IRI occurs in the document.
"""
return any(match.group() == target_url for match in IRI_PATTERN.finditer(body))
def ensure_public_request(request: httpx.Request) -> None:
"""
Ensure that an HTTP request targets a public network address.
:param request: HTTP request to validate.
:raises VerificationError: If the URL has no hostname or resolves to a
non-public address.
:raises TemporaryFetchError: If the hostname cannot be resolved.
"""
try:
if not is_public_url(str(request.url)):
raise VerificationError("Source resolves to a non-public address")
except ValueError as exc:
raise VerificationError("Source URL has no hostname") from exc
except AddressResolutionError as exc:
raise TemporaryFetchError("Source hostname could not be resolved") from exc
def fetch_source(source_url: str) -> tuple[httpx.Response, bytes]:
"""
Fetch a webmention source document.
The request follows redirects while enforcing the configured timeout,
redirect limit, and maximum response size.
:param source_url: URL of the source document.
:return: HTTP response and response body.
:raises SourceGoneError: If the source returns HTTP 410.
:raises TemporaryFetchError: If the source returns a temporary HTTP error.
:raises VerificationError: If the source returns another unsuccessful HTTP
status or exceeds the configured maximum response size.
"""
with (
httpx.Client(
headers={
"Accept": "text/html, application/xhtml+xml;q=0.9, text/plain;q=0.8",
"User-Agent": f"{APP_NAME}/{VERSION} ReceivedWebmention",
},
timeout=httpx.Timeout(
current_app.config.get("WEBMENTIONS_SSG_REQUEST_TIMEOUT", 5.0)
),
follow_redirects=True,
max_redirects=current_app.config.get("WEBMENTIONS_SSG_MAX_REDIRECTS", 20),
trust_env=False,
event_hooks={"request": [ensure_public_request]},
) as client,
client.stream("GET", source_url) as response,
):
match response.status_code:
case 200:
pass
case 410:
raise SourceGoneError("Source returned HTTP 410")
case status:
if status in {408, 425, 429} or 500 <= status <= 599:
raise TemporaryFetchError(f"Source returned HTTP {status}")
else:
raise VerificationError(f"Source returned HTTP {status}")
max_source_bytes = current_app.config.get(
"WEBMENTIONS_SSG_MAX_SOURCE_BYTES", 1_000_000
)
if (content_length := response.headers.get("Content-Length")) is not None:
try:
if int(content_length) > max_source_bytes:
raise VerificationError("Source document is too large")
except ValueError:
pass
body = bytearray()
for chunk in response.iter_bytes(chunk_size=64 * 1024):
body.extend(chunk)
if len(body) > max_source_bytes:
raise VerificationError("Source document is too large")
return response, bytes(body)
def source_mentions_target(source_url: str, target_url: str) -> bool:
"""
Fetch a source and check whether it mentions the target URL.
HTML and XHTML sources are inspected for supported URL attributes, while
plain-text sources are searched for the exact target IRI.
:param source_url: URL of the source document.
:param target_url: Exact target URL to look for.
:return: Whether the source mentions the target URL.
:raises VerificationError: If the source has an unsupported or missing
content type.
:raises SourceGoneError: If the source explicitly reports that it is gone.
:raises TemporaryFetchError: If fetching the source fails temporarily.
"""
response, body = fetch_source(source_url)
media_type = (
response.headers.get("Content-Type", "").partition(";")[0].strip().lower()
)
match media_type:
case "text/html" | "application/xhtml+xml":
return html_mentions_target(body, str(response.url), target_url)
case "text/plain":
try:
decoded_body = body.decode(
response.encoding or "utf-8", errors="replace"
)
except LookupError:
decoded_body = body.decode("utf-8", errors="replace")
return text_mentions_target(decoded_body, target_url)
case _:
raise VerificationError(
f"Unsupported source content type: {media_type or 'missing'}"
)
@huey.task(retries=2, retry_delay=50)
def verify_webmention(webmention_uuid: uuid.UUID) -> None:
"""
Verify a received webmention and store its verification status.
Temporary fetch failures are stored before being re-raised so that Huey can
retry the task.
:param webmention_uuid: Identifier of the received webmention to verify.
:raises TemporaryFetchError: If fetching the source fails temporarily.
:raises httpx.RequestError: If an HTTP request error occurs while fetching
the source.
"""
webmention = db.session.get(ReceivedWebmention, webmention_uuid)
if webmention is None:
current_app.logger.warning(
"Cannot verify unknown ReceivedWebmention %s", webmention_uuid
)
return
webmention.status = "verifying"
webmention.failure_reason = None
db.session.commit()
try:
mentions_target = source_mentions_target(webmention.source, webmention.target)
except SourceGoneError as exc:
webmention.status = "deleted"
webmention.failure_reason = str(exc)
except VerificationError as exc:
webmention.status = "failed"
webmention.failure_reason = str(exc)
except (TemporaryFetchError, httpx.RequestError) as exc:
webmention.status = "failed"
webmention.failure_reason = str(exc) or "Source could not be fetched"
db.session.commit()
# Huey retries the task because the exception escapes.
raise
else:
if mentions_target:
webmention.status = "verified"
webmention.failure_reason = None
else:
webmention.status = "deleted"
webmention.failure_reason = "Source does not mention target"
db.session.commit()
|