"""The checks a DTD cannot express: prologue, head, sizes, images, transport.""" import pytest from mews.fetch import Fetched from mews.lint import SIZE_MUST, validate_bytes CASES = { "no-marker": ("3.2", "must", "marker"), "two-markers": ("3.2", "must", "marker"), "marker-wrong-version": ("3.2", "must", "marker-version"), "meta-bad-name": ("3.3", "must", "meta-name"), "two-stylesheets": ("3.3", "must", "link-stylesheet-count"), "bad-feed-type": ("3.3", "must", "link-alternate-type"), "no-xml-decl": ("3.1", "must", "xml-declaration"), "bom": ("3.1", "must", "xml-declaration"), "wrong-doctype": ("3.1", "must", "doctype"), "internal-subset": ("3.1", "must", "internal-subset"), "billion-laughs": ("3.1", "must", "internal-subset"), "not-wellformed": ("3", "must", "well-formed"), "wrong-namespace": ("3.1", "must", "root-element"), "data-uri-img": ("4.2", "must", "image-data-uri"), "no-lang": ("3.1", "should", "lang"), "no-viewport": ("3.3", "should", "viewport"), } @pytest.mark.parametrize(("name", "expected"), CASES.items()) def test_rule_failure_maps_to_its_section(fail_page, name, expected): report = validate_bytes(fail_page(name)) section, level, code = expected match = [f for f in report.findings if f.code == code] assert match, f"{name}: no {code} in {[f.code for f in report.findings]}" assert (match[0].section, match[0].level) == (section, level) def test_either_quote_character_opens_the_declaration(conforming): """Serialisers differ: lxml writes single quotes, and XML allows them.""" page = conforming().replace( b'', b"", ) report = validate_bytes(page) assert "xml-declaration" not in [f.code for f in report.findings] assert report.conforms def test_entity_declarations_are_refused_before_parsing(fail_page): """A page declaring its own entities is rejected without expanding them.""" report = validate_bytes(fail_page("billion-laughs")) assert [f.code for f in report.findings] == ["internal-subset"] def test_should_level_findings_still_conform(fail_page): assert validate_bytes(fail_page("no-viewport")).conforms assert validate_bytes(fail_page("no-lang")).conforms def test_offsite_image_needs_the_page_address(fail_page): """Without a URL there is no site to compare against, so nothing is claimed.""" page = fail_page("offsite-img") assert not [f for f in validate_bytes(page).findings if f.code == "image-offsite"] report = validate_bytes(page, url="https://example.com/x.html") assert [f.code for f in report.findings if f.code == "image-offsite"] def test_subdomains_of_one_site_are_the_same_site(conforming): page = conforming(body='

a

') report = validate_bytes(page, url="https://example.com/x.html") assert not [f for f in report.findings if f.code == "image-offsite"] def test_oversized_page_is_reported_not_truncated(conforming): filler = "

" + ("x" * 200) + "

\n " page = conforming(body=filler * (SIZE_MUST // 200 + 20)) report = validate_bytes(page) assert "size-must" in [f.code for f in report.findings] assert not report.conforms def test_page_between_the_two_size_limits_only_warns(conforming): filler = "

" + ("x" * 200) + "

\n " page = conforming(body=filler * 400) report = validate_bytes(page) assert "size-should" in [f.code for f in report.findings] assert report.conforms def _response(url="https://example.com/", headers=None, status=200): import httpx return Fetched(url, status, httpx.Headers(headers or {}), b"", False) def test_a_cookie_on_the_page_is_a_failure(conforming): report = validate_bytes( conforming(), url="https://example.com/", response=_response( headers={"set-cookie": "a=1", "etag": '"x"', "content-type": "text/html"} ), ) cookie = [f for f in report.findings if f.code == "cookie"] assert cookie and cookie[0].section == "7.4" assert not report.conforms def test_missing_cache_validators_warn(conforming): report = validate_bytes( conforming(), url="https://example.com/", response=_response(headers={"content-type": "text/html"}), ) assert "validators" in [f.code for f in report.findings] assert report.conforms def test_plain_http_warns(conforming): report = validate_bytes( conforming(), url="http://example.com/", response=_response( url="http://example.com/", headers={"content-type": "text/html", "etag": '"x"'}, ), ) assert "https" in [f.code for f in report.findings] def test_wrong_content_type_warns(conforming): report = validate_bytes( conforming(), url="https://example.com/", response=_response(headers={"content-type": "text/plain", "etag": '"x"'}), ) assert "content-type" in [f.code for f in report.findings] def test_title_and_description_are_picked_up(conforming): report = validate_bytes(conforming()) assert report.title == "Test site" assert report.description == "A test page." assert report.language == "en" def test_control_characters_are_stripped_from_the_title(): page = """ Good‮evil site

x

""".encode() report = validate_bytes(page) assert report.title == "Goodevil site"