diff options
| author | Dennis Fink | 2026-08-19 19:49:55 +0200 |
|---|---|---|
| committer | Dennis Fink | 2026-08-19 19:49:55 +0200 |
| commit | d0e245d63e014d268310976fc45429f08d2cbbe9 (patch) | |
| tree | ef2e96570723190baf02f05c2c4644a19ff1af78 /webmentions_ssg/tasks/scanner.py | |
| parent | d4a03623544aac0ce911d8471ff4b033bc3255e1 (diff) | |
| download | webmentions-ssg-d0e245d63e014d268310976fc45429f08d2cbbe9.tar.gz webmentions-ssg-d0e245d63e014d268310976fc45429f08d2cbbe9.zip | |
refactor(core): improve typing and test coverage
Restructure database and user CLI commands, make passwords write-only
model properties, and use the UTC datetime constant throughout the
application.
Add PEP 287-style documentation and expand receiver, scanner, sender,
and URL security tests to cover error paths and edge cases.
Diffstat (limited to 'webmentions_ssg/tasks/scanner.py')
| -rw-r--r-- | webmentions_ssg/tasks/scanner.py | 139 |
1 files changed, 127 insertions, 12 deletions
diff --git a/webmentions_ssg/tasks/scanner.py b/webmentions_ssg/tasks/scanner.py index ff75697..fa6a47b 100644 --- a/webmentions_ssg/tasks/scanner.py +++ b/webmentions_ssg/tasks/scanner.py @@ -36,7 +36,13 @@ class ScannedSource: def source_url_for_path(relative_path: Path, base_url: str) -> str: - """Derive the public source URL from its relative filesystem path.""" + """ + Derive the public source URL from a relative filesystem path. + + :param relative_path: Path of the source document relative to the source root. + :param base_url: Base URL under which source documents are published. + :return: Public URL corresponding to the source document. + """ directory = relative_path.parent.as_posix() return urljoin(base_url, f"{quote(directory, safe='/')}/") @@ -45,7 +51,18 @@ def source_url_for_path(relative_path: Path, base_url: str) -> str: def canonical_url( document: BeautifulSoup, *, relative_path: Path, base_url: str | None ) -> str | None: - """Return the canonical URL declared by the document.""" + """ + Return the canonical URL declared by a source document. + + Relative canonical URLs are resolved against the configured source base URL. + + :param document: Parsed HTML source document. + :param relative_path: Path of the source document relative to the source root. + :param base_url: Configured source base URL, if available. + :return: Canonical URL, or ``None`` if none is declared. + :raises SourceScanError: If the canonical URL is empty, cannot be resolved, + or does not resolve to an HTTP or HTTPS URL. + """ if (link := document.select_one('link[rel~="canonical"][href]')) is not None: href = link.get("href") @@ -72,7 +89,13 @@ def canonical_url( def property_urls(entry: dict[str, Any], property_name: str) -> Iterator[str]: - """Yield URL values from a microformats property.""" + """ + Yield URL values from a microformats property. + + :param entry: Parsed microformats entry. + :param property_name: Name of the property to inspect. + :return: Iterator over string values of the property. + """ properties = entry.get("properties") if not isinstance(properties, dict): @@ -87,7 +110,14 @@ def property_urls(entry: dict[str, Any], property_name: str) -> Iterator[str]: def parse_entry(element: Tag, base_url: str) -> dict[str, Any]: - """Parse the source h-entry with mf2py.""" + """ + Parse and validate an h-entry with mf2py. + + :param element: HTML element containing the h-entry. + :param base_url: Base URL used when parsing the microformats data. + :return: Parsed h-entry. + :raises SourceScanError: If parsing does not produce exactly one valid h-entry. + """ parsed = mf2py.parse(doc=str(element), url=base_url) if not isinstance(parsed, dict): @@ -114,7 +144,18 @@ def parse_entry(element: Tag, base_url: str) -> dict[str, Any]: def primary_entry( document: BeautifulSoup, source_url: str ) -> tuple[Tag, dict[str, Any]]: - """Return and validate the source h-entry.""" + """ + Return and validate the primary h-entry of a source document. + + The document must contain exactly one h-entry whose ``u-url`` matches the + source URL. + + :param document: Parsed HTML source document. + :param source_url: Public URL of the source document. + :return: h-entry element and its parsed microformats representation. + :raises SourceScanError: If the document does not contain exactly one h-entry + or its ``u-url`` does not match the source URL. + """ entries = document.find_all(class_="h-entry") if len(entries) != 1: @@ -134,7 +175,14 @@ def primary_entry( def content_element(entry: Tag) -> Tag: - """Return the source h-entry's e-content element.""" + """ + Return the e-content element of an h-entry. + + :param entry: HTML element containing the h-entry. + :return: The h-entry's e-content element. + :raises SourceScanError: If the h-entry does not contain exactly one + e-content element. + """ contents = entry.find_all(class_="e-content") if len(contents) != 1: @@ -144,7 +192,18 @@ def content_element(entry: Tag) -> Tag: def normalize_target(value: str, *, base_url: str, source_url: str) -> str | None: - """Resolve and validate a possible Webmention target URL.""" + """ + Resolve and validate a possible Webmention target URL. + + Empty values, non-HTTP URLs, and URLs referring to the source itself are + discarded. + + :param value: URL reference to normalize. + :param base_url: Base URL against which relative references are resolved. + :param source_url: URL of the source document. + :return: Normalized target URL, or ``None`` if the value is not a valid + Webmention target. + """ value = value.strip() if not value: @@ -169,7 +228,20 @@ def iter_targets( source_url: str, ignored_hostnames: tuple[str, ...] = (), ) -> Iterator[str]: - """Yield outgoing Webmention targets from an h-entry.""" + """ + Yield outgoing Webmention targets from an h-entry. + + Targets are collected from links in the entry content and supported + microformats reaction properties. Invalid, self-referencing, and ignored + targets are excluded. + + :param content: e-content element of the h-entry. + :param mf2_entry: Parsed microformats representation of the h-entry. + :param base_url: Base URL against which relative targets are resolved. + :param source_url: URL of the source document. + :param ignored_hostnames: Hostname patterns whose targets should be ignored. + :return: Iterator over outgoing Webmention target URLs. + """ hrefs = ( href for element in content.find_all("a", href=True) @@ -195,7 +267,18 @@ def scan_source_file( base_url: str | None, ignored_hostnames: tuple[str, ...] = (), ) -> ScannedSource: - """Parse one generated source document.""" + """ + Scan one generated source document for outgoing Webmentions. + + :param path: Path of the generated HTML document. + :param root: Root directory containing generated source documents. + :param base_url: Configured source base URL, if available. + :param ignored_hostnames: Hostname patterns whose targets should be ignored. + :return: Scanned source metadata and discovered targets. + :raises SourceScanError: If the document cannot be safely interpreted as a + Webmention source. + :raises OSError: If the source document cannot be read. + """ document = BeautifulSoup(path.read_bytes(), "html5lib") relative_path = path.relative_to(root) @@ -232,7 +315,16 @@ def scan_source_file( def create_source(scanned: ScannedSource, scan_time: datetime) -> Source: - """Create a source from a newly discovered document.""" + """ + Create a source model from a newly discovered document. + + A sent Webmention is created for each discovered target at the initial source + revision. + + :param scanned: Scanned source data. + :param scan_time: Time at which the source was discovered. + :return: Newly created source model. + """ source = Source( path=scanned.path, url=scanned.url, @@ -254,7 +346,19 @@ def create_source(scanned: ScannedSource, scan_time: datetime) -> Source: def update_source( source: Source, scanned: ScannedSource, scan_time: datetime ) -> Source: - """Update a source from a newly scanned revision.""" + """ + Update an existing source from newly scanned data. + + A changed or previously deleted source receives a new revision. Existing sent + Webmentions are updated to reflect the current targets, and newly discovered + targets are added. + + :param source: Existing source model to update. + :param scanned: Newly scanned source data. + :param scan_time: Time at which the source was scanned. + :return: Updated source model. + :raises SourceScanError: If the public URL of the source has changed. + """ if source.url != scanned.url: raise SourceScanError( f"Source path {source.path!r} changed public URL " @@ -293,7 +397,15 @@ def update_source( @huey.lock_task("scan-webmention-sources") def scan_sources() -> None: - """Scan generated source documents and queue pending Webmentions.""" + """ + Scan generated source documents and queue pending Webmentions. + + New and changed sources are persisted, missing sources are marked as deleted, + and Webmentions requiring processing are queued for sending. + + :raises RuntimeError: If the configured source directory or base URL is + invalid. + """ directory = current_app.config.get("WEBMENTIONS_SSG_SOURCE_DIRECTORY") if directory is None: @@ -399,4 +511,7 @@ def scan_sources() -> None: @huey.task() def manual_scan_sources() -> None: + """ + Run a source scan as a Huey task. + """ return scan_sources() |
