mews.page/tests/test_lint_rules.py

159 lines
5.7 KiB
Python
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""The checks a DTD cannot express: prologue, head, sizes, images, transport."""
import pytest
from mews.fetch import Fetched
from mews.lint import SIZE_MUST, validate_bytes
CASES = {
"no-marker": ("3.2", "must", "marker"),
"two-markers": ("3.2", "must", "marker"),
"marker-wrong-version": ("3.2", "must", "marker-version"),
"meta-bad-name": ("3.3", "must", "meta-name"),
"two-stylesheets": ("3.3", "must", "link-stylesheet-count"),
"bad-feed-type": ("3.3", "must", "link-alternate-type"),
"no-xml-decl": ("3.1", "must", "xml-declaration"),
"bom": ("3.1", "must", "xml-declaration"),
"wrong-doctype": ("3.1", "must", "doctype"),
"internal-subset": ("3.1", "must", "internal-subset"),
"billion-laughs": ("3.1", "must", "internal-subset"),
"not-wellformed": ("3", "must", "well-formed"),
"wrong-namespace": ("3.1", "must", "root-element"),
"data-uri-img": ("4.2", "must", "image-data-uri"),
"no-lang": ("3.1", "should", "lang"),
"no-viewport": ("3.3", "should", "viewport"),
}
@pytest.mark.parametrize(("name", "expected"), CASES.items())
def test_rule_failure_maps_to_its_section(fail_page, name, expected):
report = validate_bytes(fail_page(name))
section, level, code = expected
match = [f for f in report.findings if f.code == code]
assert match, f"{name}: no {code} in {[f.code for f in report.findings]}"
assert (match[0].section, match[0].level) == (section, level)
def test_either_quote_character_opens_the_declaration(conforming):
"""Serialisers differ: lxml writes single quotes, and XML allows them."""
page = conforming().replace(
b'<?xml version="1.0" encoding="UTF-8"?>',
b"<?xml version='1.0' encoding='UTF-8'?>",
)
report = validate_bytes(page)
assert "xml-declaration" not in [f.code for f in report.findings]
assert report.conforms
def test_entity_declarations_are_refused_before_parsing(fail_page):
"""A page declaring its own entities is rejected without expanding them."""
report = validate_bytes(fail_page("billion-laughs"))
assert [f.code for f in report.findings] == ["internal-subset"]
def test_should_level_findings_still_conform(fail_page):
assert validate_bytes(fail_page("no-viewport")).conforms
assert validate_bytes(fail_page("no-lang")).conforms
def test_offsite_image_needs_the_page_address(fail_page):
"""Without a URL there is no site to compare against, so nothing is claimed."""
page = fail_page("offsite-img")
assert not [f for f in validate_bytes(page).findings if f.code == "image-offsite"]
report = validate_bytes(page, url="https://example.com/x.html")
assert [f.code for f in report.findings if f.code == "image-offsite"]
def test_subdomains_of_one_site_are_the_same_site(conforming):
page = conforming(body='<p><img src="https://img.example.com/a.png" alt="a" /></p>')
report = validate_bytes(page, url="https://example.com/x.html")
assert not [f for f in report.findings if f.code == "image-offsite"]
def test_oversized_page_is_reported_not_truncated(conforming):
filler = "<p>" + ("x" * 200) + "</p>\n "
page = conforming(body=filler * (SIZE_MUST // 200 + 20))
report = validate_bytes(page)
assert "size-must" in [f.code for f in report.findings]
assert not report.conforms
def test_page_between_the_two_size_limits_only_warns(conforming):
filler = "<p>" + ("x" * 200) + "</p>\n "
page = conforming(body=filler * 400)
report = validate_bytes(page)
assert "size-should" in [f.code for f in report.findings]
assert report.conforms
def _response(url="https://example.com/", headers=None, status=200):
import httpx
return Fetched(url, status, httpx.Headers(headers or {}), b"", False)
def test_a_cookie_on_the_page_is_a_failure(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(
headers={"set-cookie": "a=1", "etag": '"x"', "content-type": "text/html"}
),
)
cookie = [f for f in report.findings if f.code == "cookie"]
assert cookie and cookie[0].section == "7.4"
assert not report.conforms
def test_missing_cache_validators_warn(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(headers={"content-type": "text/html"}),
)
assert "validators" in [f.code for f in report.findings]
assert report.conforms
def test_plain_http_warns(conforming):
report = validate_bytes(
conforming(),
url="http://example.com/",
response=_response(
url="http://example.com/",
headers={"content-type": "text/html", "etag": '"x"'},
),
)
assert "https" in [f.code for f in report.findings]
def test_wrong_content_type_warns(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(headers={"content-type": "text/plain", "etag": '"x"'}),
)
assert "content-type" in [f.code for f in report.findings]
def test_title_and_description_are_picked_up(conforming):
report = validate_bytes(conforming())
assert report.title == "Test site"
assert report.description == "A test page."
assert report.language == "en"
def test_control_characters_are_stripped_from_the_title():
page = """<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Good‮evil site</title>
<meta name="mews-profile" content="0.1" />
</head>
<body><p>x</p></body>
</html>
""".encode()
report = validate_bytes(page)
assert report.title == "Goodevil site"