diff options
Diffstat (limited to '')
| -rw-r--r-- | tests/tasks/test_scanner.py | 1558 |
1 files changed, 1558 insertions, 0 deletions
diff --git a/tests/tasks/test_scanner.py b/tests/tasks/test_scanner.py new file mode 100644 index 0000000..65031b4 --- /dev/null +++ b/tests/tasks/test_scanner.py @@ -0,0 +1,1558 @@ +from pathlib import Path +from types import ModuleType +from uuid import UUID + +import pytest +import sqlalchemy as sa +from bs4 import BeautifulSoup +from flask import Flask +from pytest import MonkeyPatch + +from webmentions_ssg import DATABASE as db +from webmentions_ssg.models import SentWebmention, SentWebmentionStatus, Source + +BASE_URL = "https://dennisfink.me/blog/" +SOURCE_URL = f"{BASE_URL}example/" + + +@pytest.fixture +def scanner_module(app: Flask) -> ModuleType: + # Importing scanner registers Huey tasks. The dependency on the + # app fixture guarantees that Huey has been initialized first. + _ = app + + from webmentions_ssg.tasks import scanner + + return scanner + + +def run_scan(app: Flask, scanner_module: ModuleType) -> None: + with app.app_context(): + scanner_module.scan_sources() + + +def parse_html(html: str) -> BeautifulSoup: + return BeautifulSoup(html, "html5lib") + + +def write_post( + root: Path, + slug: str, + content: str, + *, + canonical: str | None = None, + entry_url: str | None = None, +) -> Path: + directory = root / slug + directory.mkdir(parents=True, exist_ok=True) + + source_url = f"{BASE_URL}{slug}/" + + if canonical is None: + canonical = source_url + + if entry_url is None: + entry_url = source_url + + path = directory / "index.html" + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{canonical}" + > + </head> + <body> + <article class="h-entry"> + <a + class="u-url" + href="{entry_url}" + > + Permalink + </a> + + <div class="e-content"> + {content} + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + return path + + +def configure_scanner(app: Flask, root: Path, *, base_url: str | None = None) -> None: + app.config["WEBMENTIONS_SSG_SOURCE_DIRECTORY"] = str(root) + app.config["WEBMENTIONS_SSG_SOURCE_BASE_URL"] = base_url + + +# --------------------------------------------------------------------------- +# Microformats parsing +# --------------------------------------------------------------------------- + + +def test_parse_entry_returns_microformats_entry(scanner_module: ModuleType) -> None: + document = parse_html( + f""" + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <div class="e-content"> + Hello world + </div> + </article> + """ + ) + + element = document.find(class_="h-entry") + + assert element is not None + + entry = scanner_module.parse_entry(element, SOURCE_URL) + + assert "h-entry" in entry["type"] + assert entry["properties"]["url"] == [SOURCE_URL] + + +def test_parse_entry_rejects_non_entry(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <article> + <p>Hello world</p> + </article> + """ + ) + + element = document.find("article") + + assert element is not None + + with pytest.raises( + scanner_module.SourceScanError, match="Expected exactly one parsed h-entry" + ): + scanner_module.parse_entry(element, SOURCE_URL) + + +def test_primary_entry_returns_source_entry(scanner_module: ModuleType) -> None: + document = parse_html( + f""" + <article class="h-entry" id="source"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <p class="p-name"> + Example post + </p> + + <div class="e-content"> + Hello world + </div> + </article> + """ + ) + + element, entry = scanner_module.primary_entry(document, SOURCE_URL) + + assert element["id"] == "source" + assert entry["properties"]["name"] == ["Example post"] + assert entry["properties"]["url"] == [SOURCE_URL] + + +def test_primary_entry_rejects_missing_h_entry(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <main> + <p>No microformats here.</p> + </main> + """ + ) + + with pytest.raises( + scanner_module.SourceScanError, match="Expected exactly one h-entry, found 0" + ): + scanner_module.primary_entry(document, SOURCE_URL) + + +def test_primary_entry_rejects_multiple_h_entries(scanner_module: ModuleType) -> None: + document = parse_html( + f""" + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + First + </a> + </article> + + <article class="h-entry"> + <a + class="u-url" + href="https://example.com/other/" + > + Second + </a> + </article> + """ + ) + + with pytest.raises( + scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2" + ): + scanner_module.primary_entry(document, SOURCE_URL) + + +def test_primary_entry_rejects_nested_h_entry(scanner_module: ModuleType) -> None: + document = parse_html( + f""" + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <div class="e-content"> + <article class="h-entry"> + <a + class="u-url" + href="https://example.com/nested/" + > + Nested + </a> + </article> + </div> + </article> + """ + ) + + with pytest.raises( + scanner_module.SourceScanError, match="Expected exactly one h-entry, found 2" + ): + scanner_module.primary_entry(document, SOURCE_URL) + + +def test_primary_entry_rejects_missing_u_url(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <article class="h-entry"> + <div class="e-content"> + Hello world + </div> + </article> + """ + ) + + with pytest.raises(scanner_module.SourceScanError, match="does not match"): + scanner_module.primary_entry(document, SOURCE_URL) + + +def test_primary_entry_rejects_nonmatching_u_url(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <article class="h-entry"> + <a + class="u-url" + href="https://example.com/other/" + > + Other + </a> + + <div class="e-content"> + Hello world + </div> + </article> + """ + ) + + with pytest.raises(scanner_module.SourceScanError, match="does not match"): + scanner_module.primary_entry(document, SOURCE_URL) + + +# --------------------------------------------------------------------------- +# e-content +# --------------------------------------------------------------------------- + + +def test_content_element_returns_e_content(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <article class="h-entry"> + <div class="e-content" id="content"> + Hello world + </div> + </article> + """ + ) + + entry = document.find(class_="h-entry") + + assert entry is not None + + content = scanner_module.content_element(entry) + + assert content["id"] == "content" + + +def test_content_element_rejects_missing_e_content(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <article class="h-entry"> + <p>Hello world</p> + </article> + """ + ) + + entry = document.find(class_="h-entry") + + assert entry is not None + + with pytest.raises( + scanner_module.SourceScanError, match="Expected exactly one e-content, found 0" + ): + scanner_module.content_element(entry) + + +def test_content_element_rejects_multiple_e_content(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <article class="h-entry"> + <div class="e-content"> + First + </div> + + <div class="e-content"> + Second + </div> + </article> + """ + ) + + entry = document.find(class_="h-entry") + + assert entry is not None + + with pytest.raises( + scanner_module.SourceScanError, match="Expected exactly one e-content, found 2" + ): + scanner_module.content_element(entry) + + +# --------------------------------------------------------------------------- +# Canonical URLs +# --------------------------------------------------------------------------- + + +def test_canonical_url_returns_absolute_url(scanner_module: ModuleType) -> None: + document = parse_html( + f""" + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + </html> + """ + ) + + result = scanner_module.canonical_url( + document, relative_path=Path("example/index.html"), base_url=None + ) + + assert result == SOURCE_URL + + +def test_canonical_url_returns_none_when_missing(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <html> + <head> + <title>Example</title> + </head> + </html> + """ + ) + + result = scanner_module.canonical_url( + document, relative_path=Path("example/index.html"), base_url=None + ) + + assert result is None + + +def test_canonical_url_resolves_relative_url_with_base_url( + scanner_module: ModuleType, +) -> None: + document = parse_html( + """ + <html> + <head> + <link + rel="canonical" + href="./canonical/" + > + </head> + </html> + """ + ) + + result = scanner_module.canonical_url( + document, relative_path=Path("example/index.html"), base_url=BASE_URL + ) + + assert result == ("https://dennisfink.me/blog/example/canonical/") + + +def test_relative_canonical_requires_base_url(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <html> + <head> + <link + rel="canonical" + href="./canonical/" + > + </head> + </html> + """ + ) + + with pytest.raises(scanner_module.SourceScanError, match="relative canonical"): + scanner_module.canonical_url( + document, relative_path=Path("example/index.html"), base_url=None + ) + + +def test_empty_canonical_is_rejected(scanner_module: ModuleType) -> None: + document = parse_html( + """ + <html> + <head> + <link + rel="canonical" + href=" " + > + </head> + </html> + """ + ) + + with pytest.raises(scanner_module.SourceScanError, match="empty canonical URL"): + scanner_module.canonical_url( + document, relative_path=Path("example/index.html"), base_url=BASE_URL + ) + + +# --------------------------------------------------------------------------- +# Target extraction +# --------------------------------------------------------------------------- + + +def test_scan_source_uses_canonical_without_base_url( + scanner_module: ModuleType, tmp_path: Path +) -> None: + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/target"> + Target + </a> + """, + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.path == "example/index.html" + assert scanned.url == SOURCE_URL + assert scanned.targets == frozenset({"https://example.com/target"}) + assert len(scanned.content_hash) == 32 + + +def test_scan_source_falls_back_to_base_url( + scanner_module: ModuleType, tmp_path: Path +) -> None: + directory = tmp_path / "example" + directory.mkdir() + + path = directory / "index.html" + path.write_text( + """ + <!doctype html> + <html> + <body> + <article class="h-entry"> + <a class="u-url" href="./"> + Permalink + </a> + + <div class="e-content"> + <a href="https://example.com/target"> + Target + </a> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=BASE_URL) + + assert scanned.url == SOURCE_URL + assert scanned.targets == frozenset({"https://example.com/target"}) + + +def test_scan_source_requires_canonical_or_base_url( + scanner_module: ModuleType, tmp_path: Path +) -> None: + directory = tmp_path / "example" + directory.mkdir() + + path = directory / "index.html" + path.write_text( + """ + <!doctype html> + <html> + <body> + <article class="h-entry"> + <a class="u-url" href="./"> + Permalink + </a> + + <div class="e-content"> + <p>Hello world.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + with pytest.raises(scanner_module.SourceScanError, match="no canonical URL"): + scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + +def test_scan_source_extracts_only_content_links( + scanner_module: ModuleType, tmp_path: Path +) -> None: + directory = tmp_path / "example" + directory.mkdir() + + path = directory / "index.html" + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <a href="https://example.com/metadata"> + Metadata link + </a> + + <div class="e-content"> + <a href="https://example.com/content"> + Content link + </a> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.targets == frozenset({"https://example.com/content"}) + + +def test_scan_source_extracts_relative_content_link( + scanner_module: ModuleType, tmp_path: Path +) -> None: + path = write_post( + tmp_path, + "example", + """ + <a href="../other/"> + Other post + </a> + """, + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.targets == frozenset({"https://dennisfink.me/blog/other/"}) + + +@pytest.mark.parametrize( + "property_name", ("in-reply-to", "like-of", "repost-of", "bookmark-of") +) +def test_scan_source_extracts_reaction_properties( + scanner_module: ModuleType, tmp_path: Path, property_name: str +) -> None: + directory = tmp_path / "example" + directory.mkdir() + + target = f"https://example.com/{property_name}" + + path = directory / "index.html" + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <a + class="u-{property_name}" + href="{target}" + > + Reaction target + </a> + + <div class="e-content"> + <p>Post content.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.targets == frozenset({target}) + + +def test_scan_source_extracts_anchor_from_e_content( + scanner_module: ModuleType, tmp_path: Path +) -> None: + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/target"> + Target + </a> + """, + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.targets == frozenset({"https://example.com/target"}) + + +@pytest.mark.parametrize("tag", ("link", "area")) +def test_scan_source_ignores_non_anchor_href_elements( + scanner_module: ModuleType, tmp_path: Path, tag: str +) -> None: + path = write_post( + tmp_path, + "example", + f""" + <{tag} href="https://example.com/target"> + """, + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.targets == frozenset() + + +def test_scan_source_ignores_self_fragment( + scanner_module: ModuleType, tmp_path: Path +) -> None: + path = write_post( + tmp_path, + "example", + """ + <a href="#section"> + Section + </a> + + <a href="https://example.com/target"> + Target + </a> + """, + ) + + scanned = scanner_module.scan_source_file(path, root=tmp_path, base_url=None) + + assert scanned.targets == frozenset({"https://example.com/target"}) + + +# --------------------------------------------------------------------------- +# Database reconciliation +# --------------------------------------------------------------------------- + + +def test_scan_creates_source_and_webmentions( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + + <a href="https://example.com/b"> + B + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + + assert source is not None + assert source.path == "example/index.html" + assert source.url == SOURCE_URL + assert source.revision == 1 + assert source.deleted_at is None + assert len(source.content_hash) == 32 + + webmentions = list( + db.session.scalars( + sa.select(SentWebmention).order_by(SentWebmention.target) + ) + ) + + assert [webmention.target for webmention in webmentions] == [ + "https://example.com/a", + "https://example.com/b", + ] + + assert all(webmention.active for webmention in webmentions) + assert all(webmention.desired_revision == 1 for webmention in webmentions) + assert all(webmention.processed_revision is None for webmention in webmentions) + + identifiers = {webmention.uuid for webmention in webmentions} + + assert set(queued) == identifiers + + +def test_unchanged_source_does_not_increment_revision( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + with app.app_context(): + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert webmention is not None + + webmention.processed_revision = 1 + webmention.sent_revision = 1 + webmention.status = SentWebmentionStatus.SENT + + db.session.commit() + + queued.clear() + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + + assert source is not None + assert source.revision == 1 + + assert queued == [] + + +def test_updated_source_increments_revision( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <p>Original content</p> + + <a href="https://example.com/a"> + A + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + with app.app_context(): + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert webmention is not None + + webmention.processed_revision = 1 + webmention.sent_revision = 1 + webmention.status = SentWebmentionStatus.SENT + + db.session.commit() + + queued.clear() + + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <div class="e-content"> + <p>Updated content</p> + + <a href="https://example.com/a"> + A + </a> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 2 + + assert webmention.active + assert webmention.desired_revision == 2 + assert webmention.processed_revision == 1 + assert webmention.sent_revision == 1 + assert webmention.pending + + identifier = webmention.uuid + + assert queued == [identifier] + + +def test_new_target_is_added_on_update( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <p>No links yet.</p> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + assert queued == [] + + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <div class="e-content"> + <a href="https://example.com/new"> + New + </a> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 2 + + assert webmention.target == ("https://example.com/new") + assert webmention.active + assert webmention.desired_revision == 2 + assert webmention.processed_revision is None + assert webmention.sent_revision is None + assert webmention.pending + + identifier = webmention.uuid + + assert queued == [identifier] + + +def test_removed_sent_target_is_queued_again( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/sent"> + Sent + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + with app.app_context(): + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert webmention is not None + + webmention.processed_revision = 1 + webmention.sent_revision = 1 + webmention.status = SentWebmentionStatus.SENT + + identifier = webmention.uuid + db.session.commit() + + queued.clear() + + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <div class="e-content"> + <p>The link is gone.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 2 + + assert not webmention.active + assert webmention.desired_revision == 2 + assert webmention.processed_revision == 1 + assert webmention.sent_revision == 1 + assert webmention.pending + + assert queued == [identifier] + + +def test_removed_unsent_target_is_not_queued( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/unsent"> + Unsent + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + queued.clear() + + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{SOURCE_URL}"> + Permalink + </a> + + <div class="e-content"> + <p>The link is gone.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 2 + + assert not webmention.active + assert webmention.desired_revision == 2 + assert webmention.processed_revision == 2 + assert webmention.sent_revision is None + assert not webmention.pending + + assert queued == [] + + +# --------------------------------------------------------------------------- +# Source deletion/restoration +# --------------------------------------------------------------------------- + + +def test_deleted_source_queues_previously_sent_webmention( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + with app.app_context(): + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert webmention is not None + + webmention.processed_revision = 1 + webmention.sent_revision = 1 + webmention.status = SentWebmentionStatus.SENT + + identifier = webmention.uuid + db.session.commit() + + queued.clear() + path.unlink() + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 2 + assert source.deleted_at is not None + + assert not webmention.active + assert webmention.desired_revision == 2 + assert webmention.processed_revision == 1 + assert webmention.sent_revision == 1 + assert webmention.pending + + assert queued == [identifier] + + +def test_deleted_source_does_not_queue_unsent_webmention( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + queued.clear() + path.unlink() + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 2 + assert source.deleted_at is not None + + assert not webmention.active + assert webmention.desired_revision == 2 + assert webmention.processed_revision == 2 + assert webmention.sent_revision is None + assert not webmention.pending + + assert queued == [] + + +def test_restored_source_creates_new_revision( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + with app.app_context(): + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert webmention is not None + + webmention.processed_revision = 1 + webmention.sent_revision = 1 + webmention.status = SentWebmentionStatus.SENT + + db.session.commit() + + path.unlink() + run_scan(app, scanner_module) + + queued.clear() + + write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + """, + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + webmention = db.session.scalar(sa.select(SentWebmention)) + + assert source is not None + assert webmention is not None + + assert source.revision == 3 + assert source.deleted_at is None + + assert webmention.active + assert webmention.desired_revision == 3 + assert webmention.sent_revision == 1 + assert webmention.pending + + identifier = webmention.uuid + + assert queued == [identifier] + + +# --------------------------------------------------------------------------- +# Queue recovery and invalid sources +# --------------------------------------------------------------------------- + + +def test_pending_webmention_is_requeued_on_next_scan( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path) + + write_post( + tmp_path, + "example", + """ + <a href="https://example.com/a"> + A + </a> + """, + ) + + queued: list[UUID] = [] + + monkeypatch.setattr(scanner_module, "send_webmention", queued.append) + + run_scan(app, scanner_module) + + assert len(queued) == 1 + + identifier = queued[0] + queued.clear() + + run_scan(app, scanner_module) + + assert queued == [identifier] + + +def test_invalid_known_source_is_not_deleted( + app: Flask, + scanner_module: ModuleType, + tmp_path: Path, + monkeypatch: MonkeyPatch, + caplog: pytest.LogCaptureFixture, +) -> None: + configure_scanner(app, tmp_path) + + path = write_post( + tmp_path, + "example", + """ + <p>Hello world.</p> + """, + ) + + monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None) + + run_scan(app, scanner_module) + + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <p>The h-entry disappeared.</p> + </body> + </html> + """, + encoding="utf-8", + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + + assert source is not None + assert source.revision == 1 + assert source.deleted_at is None + + assert "Could not scan source" in caplog.text + assert "Expected exactly one h-entry, found 0" in caplog.text + + +def test_scan_sources_accepts_valid_base_url( + app: Flask, scanner_module: ModuleType, tmp_path: Path, monkeypatch: MonkeyPatch +) -> None: + configure_scanner(app, tmp_path, base_url=BASE_URL) + + directory = tmp_path / "example" + directory.mkdir() + + path = directory / "index.html" + path.write_text( + """ + <!doctype html> + <html> + <body> + <article class="h-entry"> + <a class="u-url" href="./"> + Permalink + </a> + + <div class="e-content"> + <p>Hello world.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + monkeypatch.setattr(scanner_module, "send_webmention", lambda identifier: None) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + + assert source is not None + assert source.url == SOURCE_URL + + +def test_scan_sources_rejects_invalid_base_url( + app: Flask, scanner_module: ModuleType, tmp_path: Path +) -> None: + configure_scanner(app, tmp_path, base_url="not-a-url") + + with pytest.raises(RuntimeError, match="absolute HTTP or HTTPS URL"): + run_scan(app, scanner_module) + + +@pytest.mark.parametrize( + "href", + ( + "https://dennisfink.me/blog/other/", + "https://www.dennisfink.me/blog/other/", + "https://webmentions.dennisfink.me/endpoint", + "https://foo.bar.dennisfink.me/example", + ), +) +def test_scan_source_ignores_matching_hostname( + scanner_module: ModuleType, tmp_path: Path, href: str +) -> None: + path = write_post( + tmp_path, + "example", + f""" + <a href="{href}"> + Ignored + </a> + + <a href="https://example.com/target"> + External + </a> + """, + ) + + scanned = scanner_module.scan_source_file( + path, + root=tmp_path, + base_url=None, + ignored_hostnames=("dennisfink.me", "*.dennisfink.me"), + ) + + assert scanned.targets == frozenset({"https://example.com/target"}) + + +@pytest.mark.parametrize( + "href", + ( + "https://notdennisfink.me/example", + "https://dennisfink.me.example.com/example", + "https://example.com/example", + ), +) +def test_scan_source_keeps_nonmatching_hostname( + scanner_module: ModuleType, tmp_path: Path, href: str +) -> None: + path = write_post( + tmp_path, + "example", + f""" + <a href="{href}"> + Target + </a> + """, + ) + + scanned = scanner_module.scan_source_file( + path, + root=tmp_path, + base_url=None, + ignored_hostnames=("dennisfink.me", "*.dennisfink.me"), + ) + + assert scanned.targets == frozenset({href}) + + +def test_scan_source_ignores_reaction_to_matching_hostname( + scanner_module: ModuleType, tmp_path: Path +) -> None: + directory = tmp_path / "example" + directory.mkdir() + + path = directory / "index.html" + path.write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{SOURCE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a + class="u-url" + href="{SOURCE_URL}" + > + Permalink + </a> + + <a + class="u-in-reply-to" + href="https://www.dennisfink.me/blog/other/" + > + Reply target + </a> + + <div class="e-content"> + <p>Reply content.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + scanned = scanner_module.scan_source_file( + path, + root=tmp_path, + base_url=None, + ignored_hostnames=("dennisfink.me", "*.dennisfink.me"), + ) + + assert scanned.targets == frozenset() |
