aboutsummaryrefslogtreecommitdiff
path: root/webmentions_ssg/tasks/scanner.py
diff options
context:
space:
mode:
authorDennis Fink2026-08-19 19:49:55 +0200
committerDennis Fink2026-08-19 19:49:55 +0200
commitd0e245d63e014d268310976fc45429f08d2cbbe9 (patch)
treeef2e96570723190baf02f05c2c4644a19ff1af78 /webmentions_ssg/tasks/scanner.py
parentd4a03623544aac0ce911d8471ff4b033bc3255e1 (diff)
downloadwebmentions-ssg-d0e245d63e014d268310976fc45429f08d2cbbe9.tar.gz
webmentions-ssg-d0e245d63e014d268310976fc45429f08d2cbbe9.zip
refactor(core): improve typing and test coverage
Restructure database and user CLI commands, make passwords write-only model properties, and use the UTC datetime constant throughout the application. Add PEP 287-style documentation and expand receiver, scanner, sender, and URL security tests to cover error paths and edge cases.
Diffstat (limited to 'webmentions_ssg/tasks/scanner.py')
-rw-r--r--webmentions_ssg/tasks/scanner.py139
1 files changed, 127 insertions, 12 deletions
diff --git a/webmentions_ssg/tasks/scanner.py b/webmentions_ssg/tasks/scanner.py
index ff75697..fa6a47b 100644
--- a/webmentions_ssg/tasks/scanner.py
+++ b/webmentions_ssg/tasks/scanner.py
@@ -36,7 +36,13 @@ class ScannedSource:
def source_url_for_path(relative_path: Path, base_url: str) -> str:
- """Derive the public source URL from its relative filesystem path."""
+ """
+ Derive the public source URL from a relative filesystem path.
+
+ :param relative_path: Path of the source document relative to the source root.
+ :param base_url: Base URL under which source documents are published.
+ :return: Public URL corresponding to the source document.
+ """
directory = relative_path.parent.as_posix()
return urljoin(base_url, f"{quote(directory, safe='/')}/")
@@ -45,7 +51,18 @@ def source_url_for_path(relative_path: Path, base_url: str) -> str:
def canonical_url(
document: BeautifulSoup, *, relative_path: Path, base_url: str | None
) -> str | None:
- """Return the canonical URL declared by the document."""
+ """
+ Return the canonical URL declared by a source document.
+
+ Relative canonical URLs are resolved against the configured source base URL.
+
+ :param document: Parsed HTML source document.
+ :param relative_path: Path of the source document relative to the source root.
+ :param base_url: Configured source base URL, if available.
+ :return: Canonical URL, or ``None`` if none is declared.
+ :raises SourceScanError: If the canonical URL is empty, cannot be resolved,
+ or does not resolve to an HTTP or HTTPS URL.
+ """
if (link := document.select_one('link[rel~="canonical"][href]')) is not None:
href = link.get("href")
@@ -72,7 +89,13 @@ def canonical_url(
def property_urls(entry: dict[str, Any], property_name: str) -> Iterator[str]:
- """Yield URL values from a microformats property."""
+ """
+ Yield URL values from a microformats property.
+
+ :param entry: Parsed microformats entry.
+ :param property_name: Name of the property to inspect.
+ :return: Iterator over string values of the property.
+ """
properties = entry.get("properties")
if not isinstance(properties, dict):
@@ -87,7 +110,14 @@ def property_urls(entry: dict[str, Any], property_name: str) -> Iterator[str]:
def parse_entry(element: Tag, base_url: str) -> dict[str, Any]:
- """Parse the source h-entry with mf2py."""
+ """
+ Parse and validate an h-entry with mf2py.
+
+ :param element: HTML element containing the h-entry.
+ :param base_url: Base URL used when parsing the microformats data.
+ :return: Parsed h-entry.
+ :raises SourceScanError: If parsing does not produce exactly one valid h-entry.
+ """
parsed = mf2py.parse(doc=str(element), url=base_url)
if not isinstance(parsed, dict):
@@ -114,7 +144,18 @@ def parse_entry(element: Tag, base_url: str) -> dict[str, Any]:
def primary_entry(
document: BeautifulSoup, source_url: str
) -> tuple[Tag, dict[str, Any]]:
- """Return and validate the source h-entry."""
+ """
+ Return and validate the primary h-entry of a source document.
+
+ The document must contain exactly one h-entry whose ``u-url`` matches the
+ source URL.
+
+ :param document: Parsed HTML source document.
+ :param source_url: Public URL of the source document.
+ :return: h-entry element and its parsed microformats representation.
+ :raises SourceScanError: If the document does not contain exactly one h-entry
+ or its ``u-url`` does not match the source URL.
+ """
entries = document.find_all(class_="h-entry")
if len(entries) != 1:
@@ -134,7 +175,14 @@ def primary_entry(
def content_element(entry: Tag) -> Tag:
- """Return the source h-entry's e-content element."""
+ """
+ Return the e-content element of an h-entry.
+
+ :param entry: HTML element containing the h-entry.
+ :return: The h-entry's e-content element.
+ :raises SourceScanError: If the h-entry does not contain exactly one
+ e-content element.
+ """
contents = entry.find_all(class_="e-content")
if len(contents) != 1:
@@ -144,7 +192,18 @@ def content_element(entry: Tag) -> Tag:
def normalize_target(value: str, *, base_url: str, source_url: str) -> str | None:
- """Resolve and validate a possible Webmention target URL."""
+ """
+ Resolve and validate a possible Webmention target URL.
+
+ Empty values, non-HTTP URLs, and URLs referring to the source itself are
+ discarded.
+
+ :param value: URL reference to normalize.
+ :param base_url: Base URL against which relative references are resolved.
+ :param source_url: URL of the source document.
+ :return: Normalized target URL, or ``None`` if the value is not a valid
+ Webmention target.
+ """
value = value.strip()
if not value:
@@ -169,7 +228,20 @@ def iter_targets(
source_url: str,
ignored_hostnames: tuple[str, ...] = (),
) -> Iterator[str]:
- """Yield outgoing Webmention targets from an h-entry."""
+ """
+ Yield outgoing Webmention targets from an h-entry.
+
+ Targets are collected from links in the entry content and supported
+ microformats reaction properties. Invalid, self-referencing, and ignored
+ targets are excluded.
+
+ :param content: e-content element of the h-entry.
+ :param mf2_entry: Parsed microformats representation of the h-entry.
+ :param base_url: Base URL against which relative targets are resolved.
+ :param source_url: URL of the source document.
+ :param ignored_hostnames: Hostname patterns whose targets should be ignored.
+ :return: Iterator over outgoing Webmention target URLs.
+ """
hrefs = (
href
for element in content.find_all("a", href=True)
@@ -195,7 +267,18 @@ def scan_source_file(
base_url: str | None,
ignored_hostnames: tuple[str, ...] = (),
) -> ScannedSource:
- """Parse one generated source document."""
+ """
+ Scan one generated source document for outgoing Webmentions.
+
+ :param path: Path of the generated HTML document.
+ :param root: Root directory containing generated source documents.
+ :param base_url: Configured source base URL, if available.
+ :param ignored_hostnames: Hostname patterns whose targets should be ignored.
+ :return: Scanned source metadata and discovered targets.
+ :raises SourceScanError: If the document cannot be safely interpreted as a
+ Webmention source.
+ :raises OSError: If the source document cannot be read.
+ """
document = BeautifulSoup(path.read_bytes(), "html5lib")
relative_path = path.relative_to(root)
@@ -232,7 +315,16 @@ def scan_source_file(
def create_source(scanned: ScannedSource, scan_time: datetime) -> Source:
- """Create a source from a newly discovered document."""
+ """
+ Create a source model from a newly discovered document.
+
+ A sent Webmention is created for each discovered target at the initial source
+ revision.
+
+ :param scanned: Scanned source data.
+ :param scan_time: Time at which the source was discovered.
+ :return: Newly created source model.
+ """
source = Source(
path=scanned.path,
url=scanned.url,
@@ -254,7 +346,19 @@ def create_source(scanned: ScannedSource, scan_time: datetime) -> Source:
def update_source(
source: Source, scanned: ScannedSource, scan_time: datetime
) -> Source:
- """Update a source from a newly scanned revision."""
+ """
+ Update an existing source from newly scanned data.
+
+ A changed or previously deleted source receives a new revision. Existing sent
+ Webmentions are updated to reflect the current targets, and newly discovered
+ targets are added.
+
+ :param source: Existing source model to update.
+ :param scanned: Newly scanned source data.
+ :param scan_time: Time at which the source was scanned.
+ :return: Updated source model.
+ :raises SourceScanError: If the public URL of the source has changed.
+ """
if source.url != scanned.url:
raise SourceScanError(
f"Source path {source.path!r} changed public URL "
@@ -293,7 +397,15 @@ def update_source(
@huey.lock_task("scan-webmention-sources")
def scan_sources() -> None:
- """Scan generated source documents and queue pending Webmentions."""
+ """
+ Scan generated source documents and queue pending Webmentions.
+
+ New and changed sources are persisted, missing sources are marked as deleted,
+ and Webmentions requiring processing are queued for sending.
+
+ :raises RuntimeError: If the configured source directory or base URL is
+ invalid.
+ """
directory = current_app.config.get("WEBMENTIONS_SSG_SOURCE_DIRECTORY")
if directory is None:
@@ -399,4 +511,7 @@ def scan_sources() -> None:
@huey.task()
def manual_scan_sources() -> None:
+ """
+ Run a source scan as a Huey task.
+ """
return scan_sources()