aboutsummaryrefslogtreecommitdiff
path: root/tests/tasks/test_scanner.py
diff options
context:
space:
mode:
authorDennis Fink2026-08-15 20:54:53 +0200
committerDennis Fink2026-08-15 20:54:53 +0200
commite0f65d7ff582b32f1c7b5508bba8cda807aaae5f (patch)
tree6693b9fc259bf642e0392b4d8920fbca7ffd6dda /tests/tasks/test_scanner.py
parentb3e12b975064f0873bdb4d2834cae75925387d6b (diff)
downloadwebmentions-ssg-e0f65d7ff582b32f1c7b5508bba8cda807aaae5f.tar.gz
webmentions-ssg-e0f65d7ff582b32f1c7b5508bba8cda807aaae5f.zip
feat(sender): add outgoing Webmention pipeline
Scan generated h-entry documents for Webmention targets and track source revisions and delivery state in the database. Discover target endpoints, send Webmentions asynchronously, and reconcile updated, removed, restored, and deleted sources across scans. Add authenticated views for inspecting sent Webmentions and make the scanner schedule configurable.
Diffstat (limited to '')
-rw-r--r--tests/tasks/test_scanner.py1558
1 files changed, 1558 insertions, 0 deletions
diff --git a/tests/tasks/test_scanner.py b/tests/tasks/test_scanner.py
new file mode 100644
index 0000000..65031b4
--- /dev/null
+++ b/tests/tasks/test_scanner.py
@@ -0,0 +1,1558 @@
+from pathlib import Path
+from types import ModuleType
+from uuid import UUID
+
+import pytest
+import sqlalchemy as sa
+from bs4 import BeautifulSoup
+from flask import Flask
+from pytest import MonkeyPatch
+
+from webmentions_ssg import DATABASE as db
+from webmentions_ssg.models import SentWebmention, SentWebmentionStatus, Source
+
+BASE_URL = "https://dennisfink.me/blog/"
+SOURCE_URL = f"{BASE_URL}example/"
+
+
+@pytest.fixture
+def scanner_module(app: Flask) -> ModuleType:
+ # Importing scanner registers Huey tasks. The dependency on the
+ # app fixture guarantees that Huey has been initialized first.
+ _ = app
+
+ from webmentions_ssg.tasks import scanner
+
+ return scanner
+
+
+def run_scan(app: Flask, scanner_module: ModuleType) -> None:
+ with app.app_context():
+ scanner_module.scan_sources()
+
+
+def parse_html(html: str) -> BeautifulSoup:
+ return BeautifulSoup(html, "html5lib")
+
+
+def write_post(
+ root: Path,
+ slug: str,
+ content: str,
+ *,
+ canonical: str | None = None,
+ entry_url: str | None = None,
+) -> Path:
+ directory = root / slug
+ directory.mkdir(parents=True, exist_ok=True)
+
+ source_url = f"{BASE_URL}{slug}/"
+
+ if canonical is None:
+ canonical = source_url
+
+ if entry_url is None:
+ entry_url = source_url
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{canonical}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="{entry_url}"
+ >
+ Permalink
+ </a>
+
+ <div class="e-content">
+ {content}
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ return path
+
+
+def configure_scanner(app: Flask, root: Path, *, base_url: str | None = None) -> None:
+ app.config["WEBMENTIONS_SSG_SOURCE_DIRECTORY"] = str(root)
+ app.config["WEBMENTIONS_SSG_SOURCE_BASE_URL"] = base_url
+
+
+# ---------------------------------------------------------------------------
+# Microformats parsing
+# ---------------------------------------------------------------------------
+
+
+def test_parse_entry_returns_microformats_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ element = document.find(class_="h-entry")
+
+ assert element is not None
+
+ entry = scanner_module.parse_entry(element, SOURCE_URL)
+
+ assert "h-entry" in entry["type"]
+ assert entry["properties"]["url"] == [SOURCE_URL]
+
+
+def test_parse_entry_rejects_non_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article>
+ <p>Hello world</p>
+ </article>
+ """
+ )
+
+ element = document.find("article")
+
+ assert element is not None
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one parsed h-entry"
+ ):
+ scanner_module.parse_entry(element, SOURCE_URL)
+
+
+def test_primary_entry_returns_source_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry" id="source">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <p class="p-name">
+ Example post
+ </p>
+
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ element, entry = scanner_module.primary_entry(document, SOURCE_URL)
+
+ assert element["id"] == "source"
+ assert entry["properties"]["name"] == ["Example post"]
+ assert entry["properties"]["url"] == [SOURCE_URL]
+
+
+def test_primary_entry_rejects_missing_h_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <main>
+ <p>No microformats here.</p>
+ </main>
+ """
+ )
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one h-entry, found 0"
+ ):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_multiple_h_entries(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ First
+ </a>
+ </article>
+
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="https://example.com/other/"
+ >
+ Second
+ </a>
+ </article>
+ """
+ )
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2"
+ ):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_nested_h_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="https://example.com/nested/"
+ >
+ Nested
+ </a>
+ </article>
+ </div>
+ </article>
+ """
+ )
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2"
+ ):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_missing_u_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="does not match"):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_nonmatching_u_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="https://example.com/other/"
+ >
+ Other
+ </a>
+
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="does not match"):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+# ---------------------------------------------------------------------------
+# e-content
+# ---------------------------------------------------------------------------
+
+
+def test_content_element_returns_e_content(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <div class="e-content" id="content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ entry = document.find(class_="h-entry")
+
+ assert entry is not None
+
+ content = scanner_module.content_element(entry)
+
+ assert content["id"] == "content"
+
+
+def test_content_element_rejects_missing_e_content(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <p>Hello world</p>
+ </article>
+ """
+ )
+
+ entry = document.find(class_="h-entry")
+
+ assert entry is not None
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one e-content, found 0"
+ ):
+ scanner_module.content_element(entry)
+
+
+def test_content_element_rejects_multiple_e_content(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <div class="e-content">
+ First
+ </div>
+
+ <div class="e-content">
+ Second
+ </div>
+ </article>
+ """
+ )
+
+ entry = document.find(class_="h-entry")
+
+ assert entry is not None
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one e-content, found 2"
+ ):
+ scanner_module.content_element(entry)
+
+
+# ---------------------------------------------------------------------------
+# Canonical URLs
+# ---------------------------------------------------------------------------
+
+
+def test_canonical_url_returns_absolute_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ </html>
+ """
+ )
+
+ result = scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=None
+ )
+
+ assert result == SOURCE_URL
+
+
+def test_canonical_url_returns_none_when_missing(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <title>Example</title>
+ </head>
+ </html>
+ """
+ )
+
+ result = scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=None
+ )
+
+ assert result is None
+
+
+def test_canonical_url_resolves_relative_url_with_base_url(
+ scanner_module: ModuleType,
+) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="./canonical/"
+ >
+ </head>
+ </html>
+ """
+ )
+
+ result = scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=BASE_URL
+ )
+
+ assert result == ("https://dennisfink.me/blog/example/canonical/")
+
+
+def test_relative_canonical_requires_base_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="./canonical/"
+ >
+ </head>
+ </html>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="relative canonical"):
+ scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=None
+ )
+
+
+def test_empty_canonical_is_rejected(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href=" "
+ >
+ </head>
+ </html>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="empty canonical URL"):
+ scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=BASE_URL
+ )
+
+
+# ---------------------------------------------------------------------------
+# Target extraction
+# ---------------------------------------------------------------------------
+
+
+def test_scan_source_uses_canonical_without_base_url(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/target">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.path == "example/index.html"
+ assert scanned.url == SOURCE_URL
+ assert scanned.targets == frozenset({"https://example.com/target"})
+ assert len(scanned.content_hash) == 32
+
+
+def test_scan_source_falls_back_to_base_url(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ """
+ <!doctype html>
+ <html>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="./">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <a href="https://example.com/target">
+ Target
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=BASE_URL)
+
+ assert scanned.url == SOURCE_URL
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+def test_scan_source_requires_canonical_or_base_url(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ """
+ <!doctype html>
+ <html>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="./">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>Hello world.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="no canonical URL"):
+ scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+
+def test_scan_source_extracts_only_content_links(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <a href="https://example.com/metadata">
+ Metadata link
+ </a>
+
+ <div class="e-content">
+ <a href="https://example.com/content">
+ Content link
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://example.com/content"})
+
+
+def test_scan_source_extracts_relative_content_link(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="../other/">
+ Other post
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://dennisfink.me/blog/other/"})
+
+
+@pytest.mark.parametrize(
+ "property_name", ("in-reply-to", "like-of", "repost-of", "bookmark-of")
+)
+def test_scan_source_extracts_reaction_properties(
+ scanner_module: ModuleType, tmp_path: Path, property_name: str
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ target = f"https://example.com/{property_name}"
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <a
+ class="u-{property_name}"
+ href="{target}"
+ >
+ Reaction target
+ </a>
+
+ <div class="e-content">
+ <p>Post content.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({target})
+
+
+def test_scan_source_extracts_anchor_from_e_content(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/target">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+@pytest.mark.parametrize("tag", ("link", "area"))
+def test_scan_source_ignores_non_anchor_href_elements(
+ scanner_module: ModuleType, tmp_path: Path, tag: str
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ f"""
+ <{tag} href="https://example.com/target">
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset()
+
+
+def test_scan_source_ignores_self_fragment(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="#section">
+ Section
+ </a>
+
+ <a href="https://example.com/target">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+# ---------------------------------------------------------------------------
+# Database reconciliation
+# ---------------------------------------------------------------------------
+
+
+def test_scan_creates_source_and_webmentions(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+
+ <a href="https://example.com/b">
+ B
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.path == "example/index.html"
+ assert source.url == SOURCE_URL
+ assert source.revision == 1
+ assert source.deleted_at is None
+ assert len(source.content_hash) == 32
+
+ webmentions = list(
+ db.session.scalars(
+ sa.select(SentWebmention).order_by(SentWebmention.target)
+ )
+ )
+
+ assert [webmention.target for webmention in webmentions] == [
+ "https://example.com/a",
+ "https://example.com/b",
+ ]
+
+ assert all(webmention.active for webmention in webmentions)
+ assert all(webmention.desired_revision == 1 for webmention in webmentions)
+ assert all(webmention.processed_revision is None for webmention in webmentions)
+
+ identifiers = {webmention.uuid for webmention in webmentions}
+
+ assert set(queued) == identifiers
+
+
+def test_unchanged_source_does_not_increment_revision(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ db.session.commit()
+
+ queued.clear()
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.revision == 1
+
+ assert queued == []
+
+
+def test_updated_source_increments_revision(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <p>Original content</p>
+
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ db.session.commit()
+
+ queued.clear()
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>Updated content</p>
+
+ <a href="https://example.com/a">
+ A
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 1
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ identifier = webmention.uuid
+
+ assert queued == [identifier]
+
+
+def test_new_target_is_added_on_update(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <p>No links yet.</p>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ assert queued == []
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <a href="https://example.com/new">
+ New
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert webmention.target == ("https://example.com/new")
+ assert webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision is None
+ assert webmention.sent_revision is None
+ assert webmention.pending
+
+ identifier = webmention.uuid
+
+ assert queued == [identifier]
+
+
+def test_removed_sent_target_is_queued_again(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/sent">
+ Sent
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ identifier = webmention.uuid
+ db.session.commit()
+
+ queued.clear()
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>The link is gone.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 1
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ assert queued == [identifier]
+
+
+def test_removed_unsent_target_is_not_queued(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/unsent">
+ Unsent
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ queued.clear()
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>The link is gone.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 2
+ assert webmention.sent_revision is None
+ assert not webmention.pending
+
+ assert queued == []
+
+
+# ---------------------------------------------------------------------------
+# Source deletion/restoration
+# ---------------------------------------------------------------------------
+
+
+def test_deleted_source_queues_previously_sent_webmention(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ identifier = webmention.uuid
+ db.session.commit()
+
+ queued.clear()
+ path.unlink()
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+ assert source.deleted_at is not None
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 1
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ assert queued == [identifier]
+
+
+def test_deleted_source_does_not_queue_unsent_webmention(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ queued.clear()
+ path.unlink()
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+ assert source.deleted_at is not None
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 2
+ assert webmention.sent_revision is None
+ assert not webmention.pending
+
+ assert queued == []
+
+
+def test_restored_source_creates_new_revision(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ db.session.commit()
+
+ path.unlink()
+ run_scan(app, scanner_module)
+
+ queued.clear()
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 3
+ assert source.deleted_at is None
+
+ assert webmention.active
+ assert webmention.desired_revision == 3
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ identifier = webmention.uuid
+
+ assert queued == [identifier]
+
+
+# ---------------------------------------------------------------------------
+# Queue recovery and invalid sources
+# ---------------------------------------------------------------------------
+
+
+def test_pending_webmention_is_requeued_on_next_scan(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ assert len(queued) == 1
+
+ identifier = queued[0]
+ queued.clear()
+
+ run_scan(app, scanner_module)
+
+ assert queued == [identifier]
+
+
+def test_invalid_known_source_is_not_deleted(
+ app: Flask,
+ scanner_module: ModuleType,
+ tmp_path: Path,
+ monkeypatch: MonkeyPatch,
+ caplog: pytest.LogCaptureFixture,
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <p>Hello world.</p>
+ """,
+ )
+
+ monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None)
+
+ run_scan(app, scanner_module)
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <p>The h-entry disappeared.</p>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.revision == 1
+ assert source.deleted_at is None
+
+ assert "Could not scan source" in caplog.text
+ assert "Expected exactly one h-entry, found 0" in caplog.text
+
+
+def test_scan_sources_accepts_valid_base_url(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path, base_url=BASE_URL)
+
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ """
+ <!doctype html>
+ <html>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="./">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>Hello world.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.url == SOURCE_URL
+
+
+def test_scan_sources_rejects_invalid_base_url(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ configure_scanner(app, tmp_path, base_url="not-a-url")
+
+ with pytest.raises(RuntimeError, match="absolute HTTP or HTTPS URL"):
+ run_scan(app, scanner_module)
+
+
+@pytest.mark.parametrize(
+ "href",
+ (
+ "https://dennisfink.me/blog/other/",
+ "https://www.dennisfink.me/blog/other/",
+ "https://webmentions.dennisfink.me/endpoint",
+ "https://foo.bar.dennisfink.me/example",
+ ),
+)
+def test_scan_source_ignores_matching_hostname(
+ scanner_module: ModuleType, tmp_path: Path, href: str
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ f"""
+ <a href="{href}">
+ Ignored
+ </a>
+
+ <a href="https://example.com/target">
+ External
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(
+ path,
+ root=tmp_path,
+ base_url=None,
+ ignored_hostnames=("dennisfink.me", "*.dennisfink.me"),
+ )
+
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+@pytest.mark.parametrize(
+ "href",
+ (
+ "https://notdennisfink.me/example",
+ "https://dennisfink.me.example.com/example",
+ "https://example.com/example",
+ ),
+)
+def test_scan_source_keeps_nonmatching_hostname(
+ scanner_module: ModuleType, tmp_path: Path, href: str
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ f"""
+ <a href="{href}">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(
+ path,
+ root=tmp_path,
+ base_url=None,
+ ignored_hostnames=("dennisfink.me", "*.dennisfink.me"),
+ )
+
+ assert scanned.targets == frozenset({href})
+
+
+def test_scan_source_ignores_reaction_to_matching_hostname(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="{SOURCE_URL}"
+ >
+ Permalink
+ </a>
+
+ <a
+ class="u-in-reply-to"
+ href="https://www.dennisfink.me/blog/other/"
+ >
+ Reply target
+ </a>
+
+ <div class="e-content">
+ <p>Reply content.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(
+ path,
+ root=tmp_path,
+ base_url=None,
+ ignored_hostnames=("dennisfink.me", "*.dennisfink.me"),
+ )
+
+ assert scanned.targets == frozenset()