mews.page/tests/test_lint_rules.py

160 lines
5.7 KiB
Python
Raw Normal View History

"""The checks a DTD cannot express: prologue, head, sizes, images, transport."""
import pytest
from mews.fetch import Fetched
from mews.lint import SIZE_MUST, validate_bytes
CASES = {
"no-marker": ("3.2", "must", "marker"),
"two-markers": ("3.2", "must", "marker"),
"marker-wrong-version": ("3.2", "must", "marker-version"),
"meta-bad-name": ("3.3", "must", "meta-name"),
"two-stylesheets": ("3.3", "must", "link-stylesheet-count"),
"bad-feed-type": ("3.3", "must", "link-alternate-type"),
"no-xml-decl": ("3.1", "must", "xml-declaration"),
"bom": ("3.1", "must", "xml-declaration"),
"wrong-doctype": ("3.1", "must", "doctype"),
"internal-subset": ("3.1", "must", "internal-subset"),
"billion-laughs": ("3.1", "must", "internal-subset"),
"not-wellformed": ("3", "must", "well-formed"),
"wrong-namespace": ("3.1", "must", "root-element"),
"data-uri-img": ("4.2", "must", "image-data-uri"),
"no-lang": ("3.1", "should", "lang"),
"no-viewport": ("3.3", "should", "viewport"),
}
@pytest.mark.parametrize(("name", "expected"), CASES.items())
def test_rule_failure_maps_to_its_section(fail_page, name, expected):
report = validate_bytes(fail_page(name))
section, level, code = expected
match = [f for f in report.findings if f.code == code]
assert match, f"{name}: no {code} in {[f.code for f in report.findings]}"
assert (match[0].section, match[0].level) == (section, level)
def test_either_quote_character_opens_the_declaration(conforming):
"""Serialisers differ: lxml writes single quotes, and XML allows them."""
page = conforming().replace(
b'<?xml version="1.0" encoding="UTF-8"?>',
b"<?xml version='1.0' encoding='UTF-8'?>",
)
report = validate_bytes(page)
assert "xml-declaration" not in [f.code for f in report.findings]
assert report.conforms
def test_entity_declarations_are_refused_before_parsing(fail_page):
"""A page declaring its own entities is rejected without expanding them."""
report = validate_bytes(fail_page("billion-laughs"))
assert [f.code for f in report.findings] == ["internal-subset"]
def test_should_level_findings_still_conform(fail_page):
assert validate_bytes(fail_page("no-viewport")).conforms
assert validate_bytes(fail_page("no-lang")).conforms
def test_offsite_image_needs_the_page_address(fail_page):
"""Without a URL there is no site to compare against, so nothing is claimed."""
page = fail_page("offsite-img")
assert not [f for f in validate_bytes(page).findings if f.code == "image-offsite"]
report = validate_bytes(page, url="https://example.com/x.html")
assert [f.code for f in report.findings if f.code == "image-offsite"]
def test_subdomains_of_one_site_are_the_same_site(conforming):
page = conforming(body='<p><img src="https://img.example.com/a.png" alt="a" /></p>')
report = validate_bytes(page, url="https://example.com/x.html")
assert not [f for f in report.findings if f.code == "image-offsite"]
def test_oversized_page_is_reported_not_truncated(conforming):
filler = "<p>" + ("x" * 200) + "</p>\n "
page = conforming(body=filler * (SIZE_MUST // 200 + 20))
report = validate_bytes(page)
assert "size-must" in [f.code for f in report.findings]
assert not report.conforms
def test_page_between_the_two_size_limits_only_warns(conforming):
filler = "<p>" + ("x" * 200) + "</p>\n "
page = conforming(body=filler * 400)
report = validate_bytes(page)
assert "size-should" in [f.code for f in report.findings]
assert report.conforms
def _response(url="https://example.com/", headers=None, status=200):
import httpx
return Fetched(url, status, httpx.Headers(headers or {}), b"", False)
def test_a_cookie_on_the_page_is_a_failure(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(
headers={"set-cookie": "a=1", "etag": '"x"', "content-type": "text/html"}
),
)
cookie = [f for f in report.findings if f.code == "cookie"]
assert cookie and cookie[0].section == "7.4"
assert not report.conforms
def test_missing_cache_validators_warn(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(headers={"content-type": "text/html"}),
)
assert "validators" in [f.code for f in report.findings]
assert report.conforms
def test_plain_http_warns(conforming):
report = validate_bytes(
conforming(),
url="http://example.com/",
response=_response(
url="http://example.com/",
headers={"content-type": "text/html", "etag": '"x"'},
),
)
assert "https" in [f.code for f in report.findings]
def test_wrong_content_type_warns(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(headers={"content-type": "text/plain", "etag": '"x"'}),
)
assert "content-type" in [f.code for f in report.findings]
def test_title_and_description_are_picked_up(conforming):
report = validate_bytes(conforming())
assert report.title == "Test site"
assert report.description == "A test page."
assert report.language == "en"
def test_control_characters_are_stripped_from_the_title():
page = """<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Good‮evil site</title>
<meta name="mews-profile" content="0.1" />
</head>
<body><p>x</p></body>
</html>
""".encode()
report = validate_bytes(page)
assert report.title == "Goodevil site"