from pathlib import Path from types import ModuleType from uuid import UUID import pytest import sqlalchemy as sa from bs4 import BeautifulSoup from flask import Flask from pytest import MonkeyPatch from webmentions_ssg import DATABASE as db from webmentions_ssg.models import SentWebmention, SentWebmentionStatus, Source BASE_URL = "https://dennisfink.me/blog/" SOURCE_URL = f"{BASE_URL}example/" @pytest.fixture def scanner_module(app: Flask) -> ModuleType: # Importing scanner registers Huey tasks. The dependency on the # app fixture guarantees that Huey has been initialized first. _ = app from webmentions_ssg.tasks import scanner return scanner def run_scan(app: Flask, scanner_module: ModuleType) -> None: with app.app_context(): scanner_module.scan_sources() def parse_html(html: str) -> BeautifulSoup: return BeautifulSoup(html, "html5lib") def write_post( root: Path, slug: str, content: str, *, canonical: str | None = None, entry_url: str | None = None, ) -> Path: directory = root / slug directory.mkdir(parents=True, exist_ok=True) source_url = f"{BASE_URL}{slug}/" if canonical is None: canonical = source_url if entry_url is None: entry_url = source_url path = directory / "index.html" path.write_text( f"""
Permalink
{content}
""", encoding="utf-8", ) return path def configure_scanner(app: Flask, root: Path, *, base_url: str | None = None) -> None: app.config["WEBMENTIONS_SSG_SOURCE_DIRECTORY"] = str(root) app.config["WEBMENTIONS_SSG_SOURCE_BASE_URL"] = base_url # --------------------------------------------------------------------------- # Microformats parsing # --------------------------------------------------------------------------- def test_parse_entry_returns_microformats_entry(scanner_module: ModuleType) -> None: document = parse_html( f"""
Permalink
Hello world
""" ) element = document.find(class_="h-entry") assert element is not None entry = scanner_module.parse_entry(element, SOURCE_URL) assert "h-entry" in entry["type"] assert entry["properties"]["url"] == [SOURCE_URL] def test_parse_entry_rejects_non_entry(scanner_module: ModuleType) -> None: document = parse_html( """

Hello world

""" ) element = document.find("article") assert element is not None with pytest.raises( scanner_module.SourceScanError, match="Expected exactly one parsed h-entry" ): scanner_module.parse_entry(element, SOURCE_URL) def test_primary_entry_returns_source_entry(scanner_module: ModuleType) -> None: document = parse_html( f"""
Permalink

Example post

Hello world
""" ) element, entry = scanner_module.primary_entry(document, SOURCE_URL) assert element["id"] == "source" assert entry["properties"]["name"] == ["Example post"] assert entry["properties"]["url"] == [SOURCE_URL] def test_primary_entry_rejects_missing_h_entry(scanner_module: ModuleType) -> None: document = parse_html( """

No microformats here.

""" ) with pytest.raises( scanner_module.SourceScanError, match="Expected exactly one h-entry, found 0" ): scanner_module.primary_entry(document, SOURCE_URL) def test_primary_entry_rejects_multiple_h_entries(scanner_module: ModuleType) -> None: document = parse_html( f"""
First
Second
""" ) with pytest.raises( scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2" ): scanner_module.primary_entry(document, SOURCE_URL) def test_primary_entry_rejects_nested_h_entry(scanner_module: ModuleType) -> None: document = parse_html( f"""
Permalink
""" ) with pytest.raises( scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2" ): scanner_module.primary_entry(document, SOURCE_URL) def test_primary_entry_rejects_missing_u_url(scanner_module: ModuleType) -> None: document = parse_html( """
Hello world
""" ) with pytest.raises(scanner_module.SourceScanError, match="does not match"): scanner_module.primary_entry(document, SOURCE_URL) def test_primary_entry_rejects_nonmatching_u_url(scanner_module: ModuleType) -> None: document = parse_html( """
Other
Hello world
""" ) with pytest.raises(scanner_module.SourceScanError, match="does not match"): scanner_module.primary_entry(document, SOURCE_URL) # --------------------------------------------------------------------------- # e-content # --------------------------------------------------------------------------- def test_content_element_returns_e_content(scanner_module: ModuleType) -> None: document = parse_html( """
Hello world
""" ) entry = document.find(class_="h-entry") assert entry is not None content = scanner_module.content_element(entry) assert content["id"] == "content" def test_content_element_rejects_missing_e_content(scanner_module: ModuleType) -> None: document = parse_html( """

Hello world

""" ) entry = document.find(class_="h-entry") assert entry is not None with pytest.raises( scanner_module.SourceScanError, match="Expected exactly one e-content, found 0" ): scanner_module.content_element(entry) def test_content_element_rejects_multiple_e_content(scanner_module: ModuleType) -> None: document = parse_html( """
First
Second
""" ) entry = document.find(class_="h-entry") assert entry is not None with pytest.raises( scanner_module.SourceScanError, match="Expected exactly one e-content, found 2" ): scanner_module.content_element(entry) # --------------------------------------------------------------------------- # Canonical URLs # --------------------------------------------------------------------------- def test_canonical_url_returns_absolute_url(scanner_module: ModuleType) -> None: document = parse_html( f""" """ ) result = scanner_module.canonical_url( document, relative_path=Path("example/index.html"), base_url=None ) assert result == SOURCE_URL def test_canonical_url_returns_none_when_missing(scanner_module: ModuleType) -> None: document = parse_html( """ Example """ ) result = scanner_module.canonical_url( document, relative_path=Path("example/index.html"), base_url=None ) assert result is None def test_canonical_url_resolves_relative_url_with_base_url( scanner_module: ModuleType, ) -> None: document = parse_html( """ """ ) result = scanner_module.canonical_url( document, relative_path=Path("example/index.html"), base_url=BASE_URL ) assert result == ("https://dennisfink.me/blog/example/canonical/") def test_relative_canonical_requires_base_url(scanner_module: ModuleType) -> None: document = parse_html( """ """ ) with pytest.raises(scanner_module.SourceScanError, match="relative canonical"): scanner_module.canonical_url( document, relative_path=Path("example/index.html"), base_url=None ) def test_empty_canonical_is_rejected(scanner_module: ModuleType) -> None: document = parse_html( """ """ ) with pytest.raises(scanner_module.SourceScanError, match="empty canonical URL"): scanner_module.canonical_url( document, relative_path=Path("example/index.html"), base_url=BASE_URL ) # --------------------------------------------------------------------------- # Target extraction # --------------------------------------------------------------------------- def test_scan_source_uses_canonical_without_base_url( scanner_module: ModuleType, tmp_path: Path ) -> None: path = write_post( tmp_path, "example", """ Target """, ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.path == "example/index.html" assert scanned.url == SOURCE_URL assert scanned.targets == frozenset({"https://example.com/target"}) assert len(scanned.content_hash) == 32 def test_scan_source_falls_back_to_base_url( scanner_module: ModuleType, tmp_path: Path ) -> None: directory = tmp_path / "example" directory.mkdir() path = directory / "index.html" path.write_text( """
Permalink
Target
""", encoding="utf-8", ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=BASE_URL) assert scanned.url == SOURCE_URL assert scanned.targets == frozenset({"https://example.com/target"}) def test_scan_source_requires_canonical_or_base_url( scanner_module: ModuleType, tmp_path: Path ) -> None: directory = tmp_path / "example" directory.mkdir() path = directory / "index.html" path.write_text( """
Permalink

Hello world.

""", encoding="utf-8", ) with pytest.raises(scanner_module.SourceScanError, match="no canonical URL"): scanner_module.scan_source_file(path, root=tmp_path, base_url=None) def test_scan_source_extracts_only_content_links( scanner_module: ModuleType, tmp_path: Path ) -> None: directory = tmp_path / "example" directory.mkdir() path = directory / "index.html" path.write_text( f"""
Permalink Metadata link
Content link
""", encoding="utf-8", ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.targets == frozenset({"https://example.com/content"}) def test_scan_source_extracts_relative_content_link( scanner_module: ModuleType, tmp_path: Path ) -> None: path = write_post( tmp_path, "example", """ Other post """, ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.targets == frozenset({"https://dennisfink.me/blog/other/"}) @pytest.mark.parametrize( "property_name", ("in-reply-to", "like-of", "repost-of", "bookmark-of") ) def test_scan_source_extracts_reaction_properties( scanner_module: ModuleType, tmp_path: Path, property_name: str ) -> None: directory = tmp_path / "example" directory.mkdir() target = f"https://example.com/{property_name}" path = directory / "index.html" path.write_text( f"""
Permalink Reaction target

Post content.

""", encoding="utf-8", ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.targets == frozenset({target}) def test_scan_source_extracts_anchor_from_e_content( scanner_module: ModuleType, tmp_path: Path ) -> None: path = write_post( tmp_path, "example", """ Target """, ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.targets == frozenset({"https://example.com/target"}) @pytest.mark.parametrize("tag", ("link", "area")) def test_scan_source_ignores_non_anchor_href_elements( scanner_module: ModuleType, tmp_path: Path, tag: str ) -> None: path = write_post( tmp_path, "example", f""" <{tag} href="https://example.com/target"> """, ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.targets == frozenset() def test_scan_source_ignores_self_fragment( scanner_module: ModuleType, tmp_path: Path ) -> None: path = write_post( tmp_path, "example", """ Section Target """, ) scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) assert scanned.targets == frozenset({"https://example.com/target"}) # --------------------------------------------------------------------------- # Database reconciliation # --------------------------------------------------------------------------- def test_scan_creates_source_and_webmentions( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) write_post( tmp_path, "example", """ A B """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) assert source is not None assert source.path == "example/index.html" assert source.url == SOURCE_URL assert source.revision == 1 assert source.deleted_at is None assert len(source.content_hash) == 32 webmentions = list( db.session.scalars( sa.select(SentWebmention).order_by(SentWebmention.target) ) ) assert [webmention.target for webmention in webmentions] == [ "https://example.com/a", "https://example.com/b", ] assert all(webmention.active for webmention in webmentions) assert all(webmention.desired_revision == 1 for webmention in webmentions) assert all(webmention.processed_revision is None for webmention in webmentions) identifiers = {webmention.uuid for webmention in webmentions} assert set(queued) == identifiers def test_unchanged_source_does_not_increment_revision( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) write_post( tmp_path, "example", """ A """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) with app.app_context(): webmention = db.session.scalar(sa.select(SentWebmention)) assert webmention is not None webmention.processed_revision = 1 webmention.sent_revision = 1 webmention.status = SentWebmentionStatus.SENT db.session.commit() queued.clear() run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) assert source is not None assert source.revision == 1 assert queued == [] def test_updated_source_increments_revision( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """

Original content

A """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) with app.app_context(): webmention = db.session.scalar(sa.select(SentWebmention)) assert webmention is not None webmention.processed_revision = 1 webmention.sent_revision = 1 webmention.status = SentWebmentionStatus.SENT db.session.commit() queued.clear() path.write_text( f"""
Permalink

Updated content

A
""", encoding="utf-8", ) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 2 assert webmention.active assert webmention.desired_revision == 2 assert webmention.processed_revision == 1 assert webmention.sent_revision == 1 assert webmention.pending identifier = webmention.uuid assert queued == [identifier] def test_new_target_is_added_on_update( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """

No links yet.

""", ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) assert queued == [] path.write_text( f"""
Permalink
New
""", encoding="utf-8", ) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 2 assert webmention.target == ("https://example.com/new") assert webmention.active assert webmention.desired_revision == 2 assert webmention.processed_revision is None assert webmention.sent_revision is None assert webmention.pending identifier = webmention.uuid assert queued == [identifier] def test_removed_sent_target_is_queued_again( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """ Sent """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) with app.app_context(): webmention = db.session.scalar(sa.select(SentWebmention)) assert webmention is not None webmention.processed_revision = 1 webmention.sent_revision = 1 webmention.status = SentWebmentionStatus.SENT identifier = webmention.uuid db.session.commit() queued.clear() path.write_text( f"""
Permalink

The link is gone.

""", encoding="utf-8", ) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 2 assert not webmention.active assert webmention.desired_revision == 2 assert webmention.processed_revision == 1 assert webmention.sent_revision == 1 assert webmention.pending assert queued == [identifier] def test_removed_unsent_target_is_not_queued( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """ Unsent """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) queued.clear() path.write_text( f"""
Permalink

The link is gone.

""", encoding="utf-8", ) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 2 assert not webmention.active assert webmention.desired_revision == 2 assert webmention.processed_revision == 2 assert webmention.sent_revision is None assert not webmention.pending assert queued == [] # --------------------------------------------------------------------------- # Source deletion/restoration # --------------------------------------------------------------------------- def test_deleted_source_queues_previously_sent_webmention( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """ A """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) with app.app_context(): webmention = db.session.scalar(sa.select(SentWebmention)) assert webmention is not None webmention.processed_revision = 1 webmention.sent_revision = 1 webmention.status = SentWebmentionStatus.SENT identifier = webmention.uuid db.session.commit() queued.clear() path.unlink() run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 2 assert source.deleted_at is not None assert not webmention.active assert webmention.desired_revision == 2 assert webmention.processed_revision == 1 assert webmention.sent_revision == 1 assert webmention.pending assert queued == [identifier] def test_deleted_source_does_not_queue_unsent_webmention( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """ A """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) queued.clear() path.unlink() run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 2 assert source.deleted_at is not None assert not webmention.active assert webmention.desired_revision == 2 assert webmention.processed_revision == 2 assert webmention.sent_revision is None assert not webmention.pending assert queued == [] def test_restored_source_creates_new_revision( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """ A """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) with app.app_context(): webmention = db.session.scalar(sa.select(SentWebmention)) assert webmention is not None webmention.processed_revision = 1 webmention.sent_revision = 1 webmention.status = SentWebmentionStatus.SENT db.session.commit() path.unlink() run_scan(app, scanner_module) queued.clear() write_post( tmp_path, "example", """ A """, ) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) webmention = db.session.scalar(sa.select(SentWebmention)) assert source is not None assert webmention is not None assert source.revision == 3 assert source.deleted_at is None assert webmention.active assert webmention.desired_revision == 3 assert webmention.sent_revision == 1 assert webmention.pending identifier = webmention.uuid assert queued == [identifier] # --------------------------------------------------------------------------- # Queue recovery and invalid sources # --------------------------------------------------------------------------- def test_pending_webmention_is_requeued_on_next_scan( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path) write_post( tmp_path, "example", """ A """, ) queued: list[UUID] = [] monkeypatch.setattr(scanner_module, "send_webmention", queued.append) run_scan(app, scanner_module) assert len(queued) == 1 identifier = queued[0] queued.clear() run_scan(app, scanner_module) assert queued == [identifier] def test_invalid_known_source_is_not_deleted( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch, caplog: pytest.LogCaptureFixture, ) -> None: configure_scanner(app, tmp_path) path = write_post( tmp_path, "example", """

Hello world.

""", ) monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None) run_scan(app, scanner_module) path.write_text( f"""

The h-entry disappeared.

""", encoding="utf-8", ) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) assert source is not None assert source.revision == 1 assert source.deleted_at is None assert "Could not scan source" in caplog.text assert "Expected exactly one h-entry, found 0" in caplog.text def test_scan_sources_accepts_valid_base_url( app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch ) -> None: configure_scanner(app, tmp_path, base_url=BASE_URL) directory = tmp_path / "example" directory.mkdir() path = directory / "index.html" path.write_text( """
Permalink

Hello world.

""", encoding="utf-8", ) monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None) run_scan(app, scanner_module) with app.app_context(): source = db.session.scalar(sa.select(Source)) assert source is not None assert source.url == SOURCE_URL def test_scan_sources_rejects_invalid_base_url( app: Flask, scanner_module: ModuleType, tmp_path: Path ) -> None: configure_scanner(app, tmp_path, base_url="not-a-url") with pytest.raises(RuntimeError, match="absolute HTTP or HTTPS URL"): run_scan(app, scanner_module) @pytest.mark.parametrize( "href", ( "https://dennisfink.me/blog/other/", "https://www.dennisfink.me/blog/other/", "https://webmentions.dennisfink.me/endpoint", "https://foo.bar.dennisfink.me/example", ), ) def test_scan_source_ignores_matching_hostname( scanner_module: ModuleType, tmp_path: Path, href: str ) -> None: path = write_post( tmp_path, "example", f""" Ignored External """, ) scanned = scanner_module.scan_source_file( path, root=tmp_path, base_url=None, ignored_hostnames=("dennisfink.me", "*.dennisfink.me"), ) assert scanned.targets == frozenset({"https://example.com/target"}) @pytest.mark.parametrize( "href", ( "https://notdennisfink.me/example", "https://dennisfink.me.example.com/example", "https://example.com/example", ), ) def test_scan_source_keeps_nonmatching_hostname( scanner_module: ModuleType, tmp_path: Path, href: str ) -> None: path = write_post( tmp_path, "example", f""" Target """, ) scanned = scanner_module.scan_source_file( path, root=tmp_path, base_url=None, ignored_hostnames=("dennisfink.me", "*.dennisfink.me"), ) assert scanned.targets == frozenset({href}) def test_scan_source_ignores_reaction_to_matching_hostname( scanner_module: ModuleType, tmp_path: Path ) -> None: directory = tmp_path / "example" directory.mkdir() path = directory / "index.html" path.write_text( f"""
Permalink Reply target

Reply content.

""", encoding="utf-8", ) scanned = scanner_module.scan_source_file( path, root=tmp_path, base_url=None, ignored_hostnames=("dennisfink.me", "*.dennisfink.me"), ) assert scanned.targets == frozenset()