aboutsummaryrefslogtreecommitdiff
path: root/tests/tasks/test_scanner.py
diff options
context:
space:
mode:
Diffstat (limited to 'tests/tasks/test_scanner.py')
-rw-r--r--tests/tasks/test_scanner.py1558
1 files changed, 1558 insertions, 0 deletions
diff --git a/tests/tasks/test_scanner.py b/tests/tasks/test_scanner.py
new file mode 100644
index 0000000..65031b4
--- /dev/null
+++ b/tests/tasks/test_scanner.py
@@ -0,0 +1,1558 @@
+from pathlib import Path
+from types import ModuleType
+from uuid import UUID
+
+import pytest
+import sqlalchemy as sa
+from bs4 import BeautifulSoup
+from flask import Flask
+from pytest import MonkeyPatch
+
+from webmentions_ssg import DATABASE as db
+from webmentions_ssg.models import SentWebmention, SentWebmentionStatus, Source
+
+BASE_URL = "https://dennisfink.me/blog/"
+SOURCE_URL = f"{BASE_URL}example/"
+
+
+@pytest.fixture
+def scanner_module(app: Flask) -> ModuleType:
+ # Importing scanner registers Huey tasks. The dependency on the
+ # app fixture guarantees that Huey has been initialized first.
+ _ = app
+
+ from webmentions_ssg.tasks import scanner
+
+ return scanner
+
+
+def run_scan(app: Flask, scanner_module: ModuleType) -> None:
+ with app.app_context():
+ scanner_module.scan_sources()
+
+
+def parse_html(html: str) -> BeautifulSoup:
+ return BeautifulSoup(html, "html5lib")
+
+
+def write_post(
+ root: Path,
+ slug: str,
+ content: str,
+ *,
+ canonical: str | None = None,
+ entry_url: str | None = None,
+) -> Path:
+ directory = root / slug
+ directory.mkdir(parents=True, exist_ok=True)
+
+ source_url = f"{BASE_URL}{slug}/"
+
+ if canonical is None:
+ canonical = source_url
+
+ if entry_url is None:
+ entry_url = source_url
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{canonical}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="{entry_url}"
+ >
+ Permalink
+ </a>
+
+ <div class="e-content">
+ {content}
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ return path
+
+
+def configure_scanner(app: Flask, root: Path, *, base_url: str | None = None) -> None:
+ app.config["WEBMENTIONS_SSG_SOURCE_DIRECTORY"] = str(root)
+ app.config["WEBMENTIONS_SSG_SOURCE_BASE_URL"] = base_url
+
+
+# ---------------------------------------------------------------------------
+# Microformats parsing
+# ---------------------------------------------------------------------------
+
+
+def test_parse_entry_returns_microformats_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ element = document.find(class_="h-entry")
+
+ assert element is not None
+
+ entry = scanner_module.parse_entry(element, SOURCE_URL)
+
+ assert "h-entry" in entry["type"]
+ assert entry["properties"]["url"] == [SOURCE_URL]
+
+
+def test_parse_entry_rejects_non_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article>
+ <p>Hello world</p>
+ </article>
+ """
+ )
+
+ element = document.find("article")
+
+ assert element is not None
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one parsed h-entry"
+ ):
+ scanner_module.parse_entry(element, SOURCE_URL)
+
+
+def test_primary_entry_returns_source_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry" id="source">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <p class="p-name">
+ Example post
+ </p>
+
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ element, entry = scanner_module.primary_entry(document, SOURCE_URL)
+
+ assert element["id"] == "source"
+ assert entry["properties"]["name"] == ["Example post"]
+ assert entry["properties"]["url"] == [SOURCE_URL]
+
+
+def test_primary_entry_rejects_missing_h_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <main>
+ <p>No microformats here.</p>
+ </main>
+ """
+ )
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one h-entry, found 0"
+ ):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_multiple_h_entries(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ First
+ </a>
+ </article>
+
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="https://example.com/other/"
+ >
+ Second
+ </a>
+ </article>
+ """
+ )
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2"
+ ):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_nested_h_entry(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="https://example.com/nested/"
+ >
+ Nested
+ </a>
+ </article>
+ </div>
+ </article>
+ """
+ )
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2"
+ ):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_missing_u_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="does not match"):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+def test_primary_entry_rejects_nonmatching_u_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="https://example.com/other/"
+ >
+ Other
+ </a>
+
+ <div class="e-content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="does not match"):
+ scanner_module.primary_entry(document, SOURCE_URL)
+
+
+# ---------------------------------------------------------------------------
+# e-content
+# ---------------------------------------------------------------------------
+
+
+def test_content_element_returns_e_content(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <div class="e-content" id="content">
+ Hello world
+ </div>
+ </article>
+ """
+ )
+
+ entry = document.find(class_="h-entry")
+
+ assert entry is not None
+
+ content = scanner_module.content_element(entry)
+
+ assert content["id"] == "content"
+
+
+def test_content_element_rejects_missing_e_content(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <p>Hello world</p>
+ </article>
+ """
+ )
+
+ entry = document.find(class_="h-entry")
+
+ assert entry is not None
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one e-content, found 0"
+ ):
+ scanner_module.content_element(entry)
+
+
+def test_content_element_rejects_multiple_e_content(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <article class="h-entry">
+ <div class="e-content">
+ First
+ </div>
+
+ <div class="e-content">
+ Second
+ </div>
+ </article>
+ """
+ )
+
+ entry = document.find(class_="h-entry")
+
+ assert entry is not None
+
+ with pytest.raises(
+ scanner_module.SourceScanError, match="Expected exactly one e-content, found 2"
+ ):
+ scanner_module.content_element(entry)
+
+
+# ---------------------------------------------------------------------------
+# Canonical URLs
+# ---------------------------------------------------------------------------
+
+
+def test_canonical_url_returns_absolute_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ f"""
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ </html>
+ """
+ )
+
+ result = scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=None
+ )
+
+ assert result == SOURCE_URL
+
+
+def test_canonical_url_returns_none_when_missing(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <title>Example</title>
+ </head>
+ </html>
+ """
+ )
+
+ result = scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=None
+ )
+
+ assert result is None
+
+
+def test_canonical_url_resolves_relative_url_with_base_url(
+ scanner_module: ModuleType,
+) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="./canonical/"
+ >
+ </head>
+ </html>
+ """
+ )
+
+ result = scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=BASE_URL
+ )
+
+ assert result == ("https://dennisfink.me/blog/example/canonical/")
+
+
+def test_relative_canonical_requires_base_url(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="./canonical/"
+ >
+ </head>
+ </html>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="relative canonical"):
+ scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=None
+ )
+
+
+def test_empty_canonical_is_rejected(scanner_module: ModuleType) -> None:
+ document = parse_html(
+ """
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href=" "
+ >
+ </head>
+ </html>
+ """
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="empty canonical URL"):
+ scanner_module.canonical_url(
+ document, relative_path=Path("example/index.html"), base_url=BASE_URL
+ )
+
+
+# ---------------------------------------------------------------------------
+# Target extraction
+# ---------------------------------------------------------------------------
+
+
+def test_scan_source_uses_canonical_without_base_url(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/target">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.path == "example/index.html"
+ assert scanned.url == SOURCE_URL
+ assert scanned.targets == frozenset({"https://example.com/target"})
+ assert len(scanned.content_hash) == 32
+
+
+def test_scan_source_falls_back_to_base_url(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ """
+ <!doctype html>
+ <html>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="./">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <a href="https://example.com/target">
+ Target
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=BASE_URL)
+
+ assert scanned.url == SOURCE_URL
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+def test_scan_source_requires_canonical_or_base_url(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ """
+ <!doctype html>
+ <html>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="./">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>Hello world.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ with pytest.raises(scanner_module.SourceScanError, match="no canonical URL"):
+ scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+
+def test_scan_source_extracts_only_content_links(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <a href="https://example.com/metadata">
+ Metadata link
+ </a>
+
+ <div class="e-content">
+ <a href="https://example.com/content">
+ Content link
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://example.com/content"})
+
+
+def test_scan_source_extracts_relative_content_link(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="../other/">
+ Other post
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://dennisfink.me/blog/other/"})
+
+
+@pytest.mark.parametrize(
+ "property_name", ("in-reply-to", "like-of", "repost-of", "bookmark-of")
+)
+def test_scan_source_extracts_reaction_properties(
+ scanner_module: ModuleType, tmp_path: Path, property_name: str
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ target = f"https://example.com/{property_name}"
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <a
+ class="u-{property_name}"
+ href="{target}"
+ >
+ Reaction target
+ </a>
+
+ <div class="e-content">
+ <p>Post content.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({target})
+
+
+def test_scan_source_extracts_anchor_from_e_content(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/target">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+@pytest.mark.parametrize("tag", ("link", "area"))
+def test_scan_source_ignores_non_anchor_href_elements(
+ scanner_module: ModuleType, tmp_path: Path, tag: str
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ f"""
+ <{tag} href="https://example.com/target">
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset()
+
+
+def test_scan_source_ignores_self_fragment(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="#section">
+ Section
+ </a>
+
+ <a href="https://example.com/target">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None)
+
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+# ---------------------------------------------------------------------------
+# Database reconciliation
+# ---------------------------------------------------------------------------
+
+
+def test_scan_creates_source_and_webmentions(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+
+ <a href="https://example.com/b">
+ B
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.path == "example/index.html"
+ assert source.url == SOURCE_URL
+ assert source.revision == 1
+ assert source.deleted_at is None
+ assert len(source.content_hash) == 32
+
+ webmentions = list(
+ db.session.scalars(
+ sa.select(SentWebmention).order_by(SentWebmention.target)
+ )
+ )
+
+ assert [webmention.target for webmention in webmentions] == [
+ "https://example.com/a",
+ "https://example.com/b",
+ ]
+
+ assert all(webmention.active for webmention in webmentions)
+ assert all(webmention.desired_revision == 1 for webmention in webmentions)
+ assert all(webmention.processed_revision is None for webmention in webmentions)
+
+ identifiers = {webmention.uuid for webmention in webmentions}
+
+ assert set(queued) == identifiers
+
+
+def test_unchanged_source_does_not_increment_revision(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ db.session.commit()
+
+ queued.clear()
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.revision == 1
+
+ assert queued == []
+
+
+def test_updated_source_increments_revision(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <p>Original content</p>
+
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ db.session.commit()
+
+ queued.clear()
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>Updated content</p>
+
+ <a href="https://example.com/a">
+ A
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 1
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ identifier = webmention.uuid
+
+ assert queued == [identifier]
+
+
+def test_new_target_is_added_on_update(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <p>No links yet.</p>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ assert queued == []
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <a href="https://example.com/new">
+ New
+ </a>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert webmention.target == ("https://example.com/new")
+ assert webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision is None
+ assert webmention.sent_revision is None
+ assert webmention.pending
+
+ identifier = webmention.uuid
+
+ assert queued == [identifier]
+
+
+def test_removed_sent_target_is_queued_again(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/sent">
+ Sent
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ identifier = webmention.uuid
+ db.session.commit()
+
+ queued.clear()
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>The link is gone.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 1
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ assert queued == [identifier]
+
+
+def test_removed_unsent_target_is_not_queued(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/unsent">
+ Unsent
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ queued.clear()
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="{SOURCE_URL}">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>The link is gone.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 2
+ assert webmention.sent_revision is None
+ assert not webmention.pending
+
+ assert queued == []
+
+
+# ---------------------------------------------------------------------------
+# Source deletion/restoration
+# ---------------------------------------------------------------------------
+
+
+def test_deleted_source_queues_previously_sent_webmention(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ identifier = webmention.uuid
+ db.session.commit()
+
+ queued.clear()
+ path.unlink()
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+ assert source.deleted_at is not None
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 1
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ assert queued == [identifier]
+
+
+def test_deleted_source_does_not_queue_unsent_webmention(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ queued.clear()
+ path.unlink()
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 2
+ assert source.deleted_at is not None
+
+ assert not webmention.active
+ assert webmention.desired_revision == 2
+ assert webmention.processed_revision == 2
+ assert webmention.sent_revision is None
+ assert not webmention.pending
+
+ assert queued == []
+
+
+def test_restored_source_creates_new_revision(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert webmention is not None
+
+ webmention.processed_revision = 1
+ webmention.sent_revision = 1
+ webmention.status = SentWebmentionStatus.SENT
+
+ db.session.commit()
+
+ path.unlink()
+ run_scan(app, scanner_module)
+
+ queued.clear()
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+ webmention = db.session.scalar(sa.select(SentWebmention))
+
+ assert source is not None
+ assert webmention is not None
+
+ assert source.revision == 3
+ assert source.deleted_at is None
+
+ assert webmention.active
+ assert webmention.desired_revision == 3
+ assert webmention.sent_revision == 1
+ assert webmention.pending
+
+ identifier = webmention.uuid
+
+ assert queued == [identifier]
+
+
+# ---------------------------------------------------------------------------
+# Queue recovery and invalid sources
+# ---------------------------------------------------------------------------
+
+
+def test_pending_webmention_is_requeued_on_next_scan(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ write_post(
+ tmp_path,
+ "example",
+ """
+ <a href="https://example.com/a">
+ A
+ </a>
+ """,
+ )
+
+ queued: list[UUID] = []
+
+ monkeypatch.setattr(scanner_module, "send_webmention", queued.append)
+
+ run_scan(app, scanner_module)
+
+ assert len(queued) == 1
+
+ identifier = queued[0]
+ queued.clear()
+
+ run_scan(app, scanner_module)
+
+ assert queued == [identifier]
+
+
+def test_invalid_known_source_is_not_deleted(
+ app: Flask,
+ scanner_module: ModuleType,
+ tmp_path: Path,
+ monkeypatch: MonkeyPatch,
+ caplog: pytest.LogCaptureFixture,
+) -> None:
+ configure_scanner(app, tmp_path)
+
+ path = write_post(
+ tmp_path,
+ "example",
+ """
+ <p>Hello world.</p>
+ """,
+ )
+
+ monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None)
+
+ run_scan(app, scanner_module)
+
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <p>The h-entry disappeared.</p>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.revision == 1
+ assert source.deleted_at is None
+
+ assert "Could not scan source" in caplog.text
+ assert "Expected exactly one h-entry, found 0" in caplog.text
+
+
+def test_scan_sources_accepts_valid_base_url(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch
+) -> None:
+ configure_scanner(app, tmp_path, base_url=BASE_URL)
+
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ """
+ <!doctype html>
+ <html>
+ <body>
+ <article class="h-entry">
+ <a class="u-url" href="./">
+ Permalink
+ </a>
+
+ <div class="e-content">
+ <p>Hello world.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None)
+
+ run_scan(app, scanner_module)
+
+ with app.app_context():
+ source = db.session.scalar(sa.select(Source))
+
+ assert source is not None
+ assert source.url == SOURCE_URL
+
+
+def test_scan_sources_rejects_invalid_base_url(
+ app: Flask, scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ configure_scanner(app, tmp_path, base_url="not-a-url")
+
+ with pytest.raises(RuntimeError, match="absolute HTTP or HTTPS URL"):
+ run_scan(app, scanner_module)
+
+
+@pytest.mark.parametrize(
+ "href",
+ (
+ "https://dennisfink.me/blog/other/",
+ "https://www.dennisfink.me/blog/other/",
+ "https://webmentions.dennisfink.me/endpoint",
+ "https://foo.bar.dennisfink.me/example",
+ ),
+)
+def test_scan_source_ignores_matching_hostname(
+ scanner_module: ModuleType, tmp_path: Path, href: str
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ f"""
+ <a href="{href}">
+ Ignored
+ </a>
+
+ <a href="https://example.com/target">
+ External
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(
+ path,
+ root=tmp_path,
+ base_url=None,
+ ignored_hostnames=("dennisfink.me", "*.dennisfink.me"),
+ )
+
+ assert scanned.targets == frozenset({"https://example.com/target"})
+
+
+@pytest.mark.parametrize(
+ "href",
+ (
+ "https://notdennisfink.me/example",
+ "https://dennisfink.me.example.com/example",
+ "https://example.com/example",
+ ),
+)
+def test_scan_source_keeps_nonmatching_hostname(
+ scanner_module: ModuleType, tmp_path: Path, href: str
+) -> None:
+ path = write_post(
+ tmp_path,
+ "example",
+ f"""
+ <a href="{href}">
+ Target
+ </a>
+ """,
+ )
+
+ scanned = scanner_module.scan_source_file(
+ path,
+ root=tmp_path,
+ base_url=None,
+ ignored_hostnames=("dennisfink.me", "*.dennisfink.me"),
+ )
+
+ assert scanned.targets == frozenset({href})
+
+
+def test_scan_source_ignores_reaction_to_matching_hostname(
+ scanner_module: ModuleType, tmp_path: Path
+) -> None:
+ directory = tmp_path / "example"
+ directory.mkdir()
+
+ path = directory / "index.html"
+ path.write_text(
+ f"""
+ <!doctype html>
+ <html>
+ <head>
+ <link
+ rel="canonical"
+ href="{SOURCE_URL}"
+ >
+ </head>
+ <body>
+ <article class="h-entry">
+ <a
+ class="u-url"
+ href="{SOURCE_URL}"
+ >
+ Permalink
+ </a>
+
+ <a
+ class="u-in-reply-to"
+ href="https://www.dennisfink.me/blog/other/"
+ >
+ Reply target
+ </a>
+
+ <div class="e-content">
+ <p>Reply content.</p>
+ </div>
+ </article>
+ </body>
+ </html>
+ """,
+ encoding="utf-8",
+ )
+
+ scanned = scanner_module.scan_source_file(
+ path,
+ root=tmp_path,
+ base_url=None,
+ ignored_hostnames=("dennisfink.me", "*.dennisfink.me"),
+ )
+
+ assert scanned.targets == frozenset()