diff options
| author | Dennis Fink | 2026-08-19 19:49:55 +0200 |
|---|---|---|
| committer | Dennis Fink | 2026-08-19 19:49:55 +0200 |
| commit | d0e245d63e014d268310976fc45429f08d2cbbe9 (patch) | |
| tree | ef2e96570723190baf02f05c2c4644a19ff1af78 /tests/tasks/test_scanner.py | |
| parent | d4a03623544aac0ce911d8471ff4b033bc3255e1 (diff) | |
| download | webmentions-ssg-d0e245d63e014d268310976fc45429f08d2cbbe9.tar.gz webmentions-ssg-d0e245d63e014d268310976fc45429f08d2cbbe9.zip | |
refactor(core): improve typing and test coverage
Restructure database and user CLI commands, make passwords write-only
model properties, and use the UTC datetime constant throughout the
application.
Add PEP 287-style documentation and expand receiver, scanner, sender,
and URL security tests to cover error paths and edge cases.
Diffstat (limited to 'tests/tasks/test_scanner.py')
| -rw-r--r-- | tests/tasks/test_scanner.py | 203 |
1 files changed, 203 insertions, 0 deletions
diff --git a/tests/tasks/test_scanner.py b/tests/tasks/test_scanner.py index 65031b4..89e2995 100644 --- a/tests/tasks/test_scanner.py +++ b/tests/tasks/test_scanner.py @@ -1556,3 +1556,206 @@ def test_scan_source_ignores_reaction_to_matching_hostname( ) assert scanned.targets == frozenset() + + +def test_canonical_url_rejects_non_http_resolved_url( + scanner_module: ModuleType, +) -> None: + document = parse_html( + """ + <html> + <head> + <link + rel="canonical" + href="mailto:example@example.com" + > + </head> + </html> + """ + ) + + with pytest.raises( + scanner_module.SourceScanError, + match="Canonical URL is not a valid HTTP or HTTPS URL", + ): + scanner_module.canonical_url( + document, relative_path=Path("example/index.html"), base_url=BASE_URL + ) + + +def test_property_urls_ignores_non_mapping_properties( + scanner_module: ModuleType, +) -> None: + entry = {"properties": "invalid"} + + assert tuple(scanner_module.property_urls(entry, "url")) == () + + +@pytest.mark.parametrize( + ("parsed", "message"), + [ + pytest.param( + [], + "Microformats parser did not return a document object", + id="non-object-document", + ), + pytest.param( + {"items": [None]}, "Parsed h-entry is not an object", id="non-object-entry" + ), + pytest.param( + {"items": [{"type": ["h-card"]}]}, + "Parsed microformats item is not an h-entry", + id="wrong-item-type", + ), + ], +) +def test_parse_entry_rejects_invalid_parser_output( + scanner_module: ModuleType, monkeypatch: MonkeyPatch, parsed: object, message: str +) -> None: + document = parse_html( + """ + <article class="h-entry"> + <p>Hello world</p> + </article> + """ + ) + + element = document.find(class_="h-entry") + + assert element is not None + + monkeypatch.setattr(scanner_module.mf2py, "parse", lambda **kwargs: parsed) + + with pytest.raises(scanner_module.SourceScanError, match=message): + scanner_module.parse_entry(element, SOURCE_URL) + + +@pytest.mark.parametrize( + "value", + [ + pytest.param(" ", id="empty"), + pytest.param("mailto:example@example.com", id="non-http"), + ], +) +def test_normalize_target_ignores_invalid_target( + scanner_module: ModuleType, value: str +) -> None: + assert ( + scanner_module.normalize_target( + value, base_url=SOURCE_URL, source_url=SOURCE_URL + ) + is None + ) + + +def test_changed_source_url_is_rejected( + app: Flask, + scanner_module: ModuleType, + tmp_path: Path, + caplog: pytest.LogCaptureFixture, +) -> None: + configure_scanner(app, tmp_path) + + write_post( + tmp_path, + "example", + """ + <p>Hello world.</p> + """, + ) + + run_scan(app, scanner_module) + + changed_url = f"{BASE_URL}changed/" + + write_post( + tmp_path, + "example", + """ + <p>Hello world.</p> + """, + canonical=changed_url, + entry_url=changed_url, + ) + + run_scan(app, scanner_module) + + with app.app_context(): + source = db.session.scalar(sa.select(Source)) + + assert source is not None + assert source.url == SOURCE_URL + assert source.revision == 1 + + assert "changed public URL" in caplog.text + + +def test_scan_sources_returns_when_source_directory_is_not_configured( + app: Flask, scanner_module: ModuleType +) -> None: + app.config["WEBMENTIONS_SSG_SOURCE_DIRECTORY"] = None + + run_scan(app, scanner_module) + + with app.app_context(): + assert db.session.scalar(sa.select(Source)) is None + + +def test_scan_sources_rejects_source_directory_that_is_not_directory( + app: Flask, scanner_module: ModuleType, tmp_path: Path +) -> None: + path = tmp_path / "source.html" + path.write_text("<p>Hello world.</p>", encoding="utf-8") + + configure_scanner(app, path) + + with pytest.raises(RuntimeError, match="Source directory is not a directory"): + run_scan(app, scanner_module) + + +def test_scan_sources_ignores_root_index( + app: Flask, scanner_module: ModuleType, tmp_path: Path +) -> None: + configure_scanner(app, tmp_path) + + (tmp_path / "index.html").write_text( + f""" + <!doctype html> + <html> + <head> + <link + rel="canonical" + href="{BASE_URL}" + > + </head> + <body> + <article class="h-entry"> + <a class="u-url" href="{BASE_URL}"> + Permalink + </a> + + <div class="e-content"> + <p>Blog index.</p> + </div> + </article> + </body> + </html> + """, + encoding="utf-8", + ) + + write_post( + tmp_path, + "example", + """ + <p>Hello world.</p> + """, + ) + + run_scan(app, scanner_module) + + with app.app_context(): + sources = list(db.session.scalars(sa.select(Source))) + + assert len(sources) == 1 + assert sources[0].path == "example/index.html" |
