test: cover the validator, the directory and the site

This commit is contained in:
randogoth 2026-10-11 15:05:42 +03:00
parent 003bb0b4b1
commit 1e75a69381
44 changed files with 1796 additions and 0 deletions

143
tests/conftest.py Normal file
View file

@ -0,0 +1,143 @@
"""Shared fixtures: fixture paths, a temporary database, and a local web server."""
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
import threading
import pytest
from mews import db
ROOT = Path(__file__).resolve().parent.parent
FIXTURES = Path(__file__).resolve().parent / "fixtures"
@pytest.fixture
def fail_page():
"""Read one of the non-conforming fixture pages."""
def read(name: str) -> bytes:
return (FIXTURES / "fail" / f"{name}.html").read_bytes()
return read
@pytest.fixture
def connection(tmp_path):
"""An empty database."""
handle = db.connect(str(tmp_path / "mews.db"))
db.init(handle)
yield handle
handle.close()
class Site:
"""A web server on loopback whose responses the test writes."""
def __init__(self) -> None:
self.routes: dict[str, tuple[int, dict[str, str], bytes, bool]] = {}
self.requests: list[tuple[str, dict[str, str]]] = []
self.delay = 0.0
self._server = ThreadingHTTPServer(("127.0.0.1", 0), _handler(self))
self._thread = threading.Thread(target=self._server.serve_forever, daemon=True)
self._thread.start()
@property
def base(self) -> str:
"""The address the server answers on."""
host, port = self._server.server_address[:2]
return f"http://{host}:{port}"
def add(
self,
path: str,
body: bytes = b"",
status: int = 200,
headers: dict[str, str] | None = None,
declare_length: bool = True,
) -> str:
"""Serve body at path, and return its full address.
With declare_length off the response carries no Content-Length and ends
at the connection close, which is how a streaming body reaches the
fetcher's byte cap.
"""
self.routes[path] = (status, headers or {}, body, declare_length)
return self.base + path
def close(self) -> None:
"""Stop the server."""
self._server.shutdown()
self._server.server_close()
def _handler(site: "Site"):
class Handler(BaseHTTPRequestHandler):
protocol_version = "HTTP/1.1"
def do_GET(self):
import time
site.requests.append((self.path, dict(self.headers)))
if site.delay:
time.sleep(site.delay)
status, headers, body, declare_length = site.routes.get(
self.path, (404, {}, b"not here", True)
)
if not declare_length:
self.protocol_version = "HTTP/1.0"
self.close_connection = True
self.send_response(status)
for name, value in headers.items():
self.send_header(name, value)
if "Content-Type" not in headers and status == 200:
self.send_header("Content-Type", "text/html; charset=utf-8")
if declare_length:
self.send_header("Content-Length", str(len(body)))
self.end_headers()
if body:
self.wfile.write(body)
def log_message(self, *args):
pass
def handle_one_request(self):
try:
super().handle_one_request()
except (BrokenPipeError, ConnectionResetError):
self.close_connection = True
return Handler
@pytest.fixture
def site():
"""A local web server for the checks that need a live page."""
server = Site()
yield server
server.close()
@pytest.fixture
def conforming():
"""A page that follows the spec, with a stylesheet link the test can point."""
def build(stylesheet: str = "mews-0.1.css", body: str = "<p>Hello.</p>") -> bytes:
return f"""<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test site</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<meta name="description" content="A test page." />
<link rel="stylesheet" type="text/css" href="{stylesheet}" />
</head>
<body>
{body}
</body>
</html>
""".encode()
return build

14
tests/fixtures/fail/bad-body-class.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body class="mews-pink">
<p>Hello.</p>
</body>
</html>

15
tests/fixtures/fail/bad-feed-type.html vendored Normal file
View file

@ -0,0 +1,15 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="alternate" type="application/rss+xml" href="feed.xml" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

18
tests/fixtures/fail/billion-laughs.html vendored Normal file
View file

@ -0,0 +1,18 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd" [
<!ENTITY lol "lol">
<!ENTITY lol1 "&lol;&lol;&lol;&lol;&lol;&lol;&lol;&lol;&lol;&lol;">
<!ENTITY lol2 "&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;&lol1;">
]>
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>&lol2;</p>
</body>
</html>

14
tests/fixtures/fail/bom.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

14
tests/fixtures/fail/class-on-p.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p class="mews-warm">Hello.</p>
</body>
</html>

14
tests/fixtures/fail/data-uri-img.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p><img src="data:image/gif;base64,R0lGODl=" alt="x" /></p>
</body>
</html>

14
tests/fixtures/fail/div.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<div><p>Hello.</p></div>
</body>
</html>

14
tests/fixtures/fail/empty-body.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
</body>
</html>

View file

@ -0,0 +1,13 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
<meta name="mews-profile" content="0.1" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

14
tests/fixtures/fail/img-no-alt.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p><img src="a.png" /></p>
</body>
</html>

14
tests/fixtures/fail/img-width.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p><img src="a.png" alt="A square" width="10" /></p>
</body>
</html>

View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<form action="/x" method="post"><fieldset><p><input type="email" name="e" /></p></fieldset></form>
</body>
</html>

View file

@ -0,0 +1,16 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd" [
<!ENTITY boom "kaboom">
]>
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>&boom;</p>
</body>
</html>

15
tests/fixtures/fail/link-bad-rel.html vendored Normal file
View file

@ -0,0 +1,15 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="canonical" href="/x" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

View file

@ -0,0 +1,13 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="9.9" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

14
tests/fixtures/fail/meta-bad-name.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="generator" content="hand" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

14
tests/fixtures/fail/nested-table.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<table><tr><td><table><tr><td>x</td></tr></table></td></tr></table>
</body>
</html>

12
tests/fixtures/fail/no-lang.html vendored Normal file
View file

@ -0,0 +1,12 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body><p>Hello.</p></body>
</html>

13
tests/fixtures/fail/no-marker.html vendored Normal file
View file

@ -0,0 +1,13 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

13
tests/fixtures/fail/no-viewport.html vendored Normal file
View file

@ -0,0 +1,13 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

13
tests/fixtures/fail/no-xml-decl.html vendored Normal file
View file

@ -0,0 +1,13 @@
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

14
tests/fixtures/fail/not-wellformed.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.
</body>
</html>

14
tests/fixtures/fail/offsite-img.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p><img src="https://cdn.example.net/a.png" alt="x" /></p>
</body>
</html>

15
tests/fixtures/fail/script.html vendored Normal file
View file

@ -0,0 +1,15 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
<script type="text/javascript">alert(1)</script>
</body>
</html>

14
tests/fixtures/fail/style-attr.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p style="color: red">Hello.</p>
</body>
</html>

15
tests/fixtures/fail/style-element.html vendored Normal file
View file

@ -0,0 +1,15 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
<style>p { color: red }</style>
</body>
</html>

14
tests/fixtures/fail/sup.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>E = mc<sup>2</sup></p>
</body>
</html>

14
tests/fixtures/fail/table-in-li.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<ul><li><table><tr><td>x</td></tr></table></li></ul>
</body>
</html>

14
tests/fixtures/fail/two-markers.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="mews-profile" content="0.1" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

View file

@ -0,0 +1,15 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
<link rel="stylesheet" type="text/css" href="extra.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

14
tests/fixtures/fail/wrong-doctype.html vendored Normal file
View file

@ -0,0 +1,14 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Strict//EN"
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-strict.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body>
<p>Hello.</p>
</body>
</html>

View file

@ -0,0 +1,12 @@
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xml:lang="en" lang="en">
<head>
<title>Test</title>
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="mews-0.1.css" />
</head>
<body><p>Hello.</p></body>
</html>

50
tests/test_coverage.py Normal file
View file

@ -0,0 +1,50 @@
"""The validator against the spec it implements.
These are the tests that notice when the spec moves and the validator doesn't.
"""
from pathlib import Path
import re
from mews.lint import CHECKS, NOT_CHECKED
SPEC = Path(__file__).resolve().parent.parent / "doc" / "SPEC.md"
KEYWORD = re.compile(r"\b(MUST NOT|MUST|SHOULD NOT|SHOULD)\b")
HEADING = re.compile(r"^#{2,3} (\d+(?:\.\d+)?)[. ]")
def sections_with_rules() -> set[str]:
"""Every section of SPEC.md 3 to 7 that states a MUST or a SHOULD."""
found: set[str] = set()
current = ""
for line in SPEC.read_text().splitlines():
heading = HEADING.match(line)
if heading:
current = heading.group(1)
continue
if current[:1] in "34567" and KEYWORD.search(line):
found.add(current)
return found
def test_every_rule_in_the_spec_is_checked_or_listed_as_unchecked():
covered = {check.section for check in CHECKS} | set(NOT_CHECKED)
missing = sections_with_rules() - covered
assert not missing, (
f"sections {sorted(missing)} of SPEC.md state rules that mews.lint "
"neither checks nor lists in NOT_CHECKED"
)
def test_the_unchecked_list_says_why():
for section, reason in NOT_CHECKED.items():
assert len(reason) > 20, section
def test_check_codes_are_unique():
codes = [check.code for check in CHECKS]
assert len(codes) == len(set(codes))
def test_every_check_is_a_must_or_a_should():
assert {check.level for check in CHECKS} == {"must", "should"}

60
tests/test_css.py Normal file
View file

@ -0,0 +1,60 @@
"""The section 5.2 comparison: an unmodified copy, plus added @font-face rules."""
from mews import css
CANONICAL = (
__import__("pathlib").Path(__file__).resolve().parent.parent / "mews-0.1.css"
).read_text()
FONT_FACE = """@font-face {
font-family: "Atkinson Hyperlegible";
src: url(/fonts/atkinson.woff2) format("woff2");
}
"""
def test_an_unmodified_copy_passes():
assert css.compare(CANONICAL).equal
def test_line_endings_and_trailing_spaces_do_not_matter():
mangled = CANONICAL.replace("\n", " \r\n") + "\n\n\n"
assert css.compare(mangled).equal
def test_an_added_font_face_passes_and_is_read():
result = css.compare(FONT_FACE + CANONICAL)
assert result.equal
assert result.font_urls == ["/fonts/atkinson.woff2"]
def test_a_font_face_at_the_end_also_passes():
assert css.compare(CANONICAL + "\n" + FONT_FACE).equal
def test_a_changed_rule_fails_with_a_line_number():
result = css.compare(CANONICAL.replace("#faf8f3", "#ffffff"))
assert not result.equal
assert result.diff_line == 5
assert "background-color" in result.diff
def test_a_reordered_rule_fails():
lines = CANONICAL.split("\n")
swapped = "\n".join([*lines[:4], lines[5], lines[4], *lines[6:]])
assert not css.compare(swapped).equal
def test_a_removed_rule_fails():
assert not css.compare(CANONICAL.replace("hr {", "hr-off {")).equal
def test_an_import_is_noticed():
assert css.compare('@import url("x.css");\n' + CANONICAL).has_import
def test_a_font_face_inside_a_media_query_is_not_stripped():
inner = "@media screen {\n" + FONT_FACE + "}\n"
result = css.compare(CANONICAL + inner)
assert not result.equal

30
tests/test_data.py Normal file
View file

@ -0,0 +1,30 @@
"""The packaged copies of the stylesheet and the DTD, against their originals.
The validator has to find both when it is installed as a wheel with nothing but
the package on disk, so mews/data holds copies. They must not drift.
"""
from importlib import resources
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
COPIES = {
"mews-0.1.css": ROOT / "mews-0.1.css",
"mews-0.1.dtd": ROOT / "dtd" / "mews-0.1.dtd",
}
def test_packaged_copies_match_the_originals():
for name, original in COPIES.items():
packaged = resources.files("mews.data").joinpath(name).read_bytes()
assert packaged == original.read_bytes(), (
f"mews/data/{name} differs from {original}. Copy the original over it."
)
def test_the_spec_quotes_the_same_stylesheet():
"""SPEC.md 5.1 prints the stylesheet in full; it has to be the same file."""
spec = (ROOT / "doc" / "SPEC.md").read_text()
quoted = spec.split("```css\n", 1)[1].split("```", 1)[0]
assert quoted.strip() == (ROOT / "mews-0.1.css").read_text().strip()

31
tests/test_domains.py Normal file
View file

@ -0,0 +1,31 @@
"""Registrable-domain comparison, which is what the same-site rules rest on."""
import pytest
from mews.fetch import registered_domain, same_site
CASES = [
("example.com", "example.com"),
("img.example.com", "example.com"),
("a.b.example.co.uk", "example.co.uk"),
("EXAMPLE.COM", "example.com"),
("example.com.", "example.com"),
("a.github.io", "a.github.io"),
("not-a-tld", None),
("", None),
]
@pytest.mark.parametrize(("host", "expected"), CASES)
def test_registrable_domain(host, expected):
assert registered_domain(host) == expected
def test_subdomains_of_one_domain_are_one_site():
assert same_site("example.com", "img.example.com")
def test_two_sites_on_a_shared_suffix_are_not_one_site():
"""The reason this uses a public suffix list rather than counting labels."""
assert not same_site("a.github.io", "b.github.io")
assert not same_site("example.co.uk", "evil.co.uk")

175
tests/test_fetch.py Normal file
View file

@ -0,0 +1,175 @@
"""The fetcher's refusals.
The submit endpoint fetches addresses strangers type, so this is the test that
keeps the service from being used as a probe against its own machine.
"""
import pytest
from mews import fetch
from mews.fetch import Fetcher, FetchError, UrlError, normalise_url, vet_address
PRIVATE = [
"127.0.0.1",
"127.1.2.3",
"10.0.0.1",
"172.16.0.1",
"192.168.1.1",
"169.254.169.254",
"0.0.0.0",
"100.64.0.1",
"192.0.0.1",
"198.18.0.1",
"192.0.2.1",
"203.0.113.1",
"240.0.0.1",
"255.255.255.255",
"::1",
"::",
"fd00::1",
"fe80::1",
"::ffff:127.0.0.1",
"2002:7f00:1::",
"64:ff9b::7f00:1",
"2001:db8::1",
]
PUBLIC = ["93.184.216.34", "1.1.1.1", "194.68.44.29", "2606:2800:220:1::"]
REFUSED = [
"file:///etc/passwd",
"javascript:alert(1)",
"gopher://example.com/",
"http://user:pw@example.com/",
"http://example.com:22/",
"http://127.0.0.1/",
"http://2130706433/",
"http://0x7f.1/",
"http://[::1]/",
"http://localhost/",
"http://box.internal/",
"http://printer.local/",
"http://not-a-tld/",
"https://exa mple.com/",
"https://a_b.example.com/",
"",
]
@pytest.mark.parametrize("address", PRIVATE)
def test_addresses_off_the_public_internet_are_refused(address):
with pytest.raises(UrlError):
vet_address(address)
@pytest.mark.parametrize("address", PUBLIC)
def test_public_addresses_pass(address):
vet_address(address)
@pytest.mark.parametrize("url", REFUSED)
def test_addresses_the_checker_will_not_read(url):
with pytest.raises(UrlError):
normalise_url(url)
def test_a_bare_domain_becomes_https():
assert normalise_url("example.com") == "https://example.com/"
def test_the_fragment_is_dropped():
assert normalise_url("https://example.com/a#b") == "https://example.com/a"
def test_the_host_is_lowercased_and_punycoded():
assert normalise_url("https://EXAMPLE.com/") == "https://example.com/"
def test_the_extra_blocked_ranges_come_from_the_environment(monkeypatch):
"""The deployment blocks its own public address this way."""
monkeypatch.setattr(
fetch, "_EXTRA_NETS", [__import__("ipaddress").ip_network("194.68.44.28/32")]
)
with pytest.raises(UrlError):
vet_address("194.68.44.28")
vet_address("194.68.44.29")
def test_loopback_is_allowed_only_when_asked():
with pytest.raises(UrlError):
vet_address("127.0.0.1")
vet_address("127.0.0.1", allow_loopback=True)
# --- against a real server on loopback ---------------------------------
def test_a_page_is_read(site):
url = site.add("/a.html", b"<p>hi</p>")
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(url)
assert response.status == 200
assert response.body == b"<p>hi</p>"
def test_the_host_header_carries_the_name_not_the_address(site):
url = site.add("/a.html", b"x")
with Fetcher(allow_loopback=True) as fetcher:
fetcher.get(url)
_, headers = site.requests[-1]
assert headers["Host"] == url.split("//")[1].split("/")[0]
def test_a_redirect_is_followed(site):
target = site.add("/b.html", b"there")
site.add("/a.html", status=302, headers={"Location": target})
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(site.base + "/a.html")
assert response.body == b"there"
assert response.url == target
def test_a_redirect_to_a_private_address_is_refused(site):
site.add("/a.html", status=302, headers={"Location": "http://10.0.0.1/"})
with Fetcher(allow_loopback=True) as fetcher, pytest.raises(UrlError):
fetcher.get(site.base + "/a.html")
def test_a_redirect_loop_stops(site):
site.add("/a.html", status=302, headers={"Location": site.base + "/a.html"})
with Fetcher(allow_loopback=True) as fetcher, pytest.raises(FetchError):
fetcher.get(site.base + "/a.html")
def test_a_body_over_the_cap_is_cut_short(site):
url = site.add("/big.html", b"x" * 400_000, declare_length=False)
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(url)
assert response.truncated
assert len(response.body) == fetch.PAGE_CAP
def test_a_declared_length_over_the_cap_is_refused_unread(site):
url = site.add("/big.html", b"x" * 100, headers={"Content-Length": "9999999"})
with Fetcher(allow_loopback=True) as fetcher, pytest.raises(FetchError):
fetcher.get(url)
def test_a_slow_page_times_out(site):
url = site.add("/slow.html", b"x")
site.delay = 0.3
with (
Fetcher(allow_loopback=True, budget=0.05) as fetcher,
pytest.raises(FetchError),
):
fetcher.get(url)
def test_a_missing_page_is_reported_by_status(site):
with Fetcher(allow_loopback=True) as fetcher:
assert fetcher.get(site.base + "/nope.html").status == 404
def test_a_name_that_does_not_resolve_is_reported():
with Fetcher() as fetcher, pytest.raises(FetchError):
fetcher.get("https://this-name-does-not-exist.example/")

78
tests/test_lint_dtd.py Normal file
View file

@ -0,0 +1,78 @@
"""The DTD check, and the mapping from libxml2's messages to spec sections.
One test per message shape, so the mapping table in mews.lint stays pinned to
what libxml2 actually says.
"""
import pytest
from mews.lint import validate_bytes
# fixture name -> (section, level, code)
CASES = {
"script": ("4.1", "must", "dtd-element"),
"div": ("4.1", "must", "dtd-element"),
"sup": ("4.1", "must", "dtd-element"),
"style-element": ("5", "must", "dtd-styling"),
"style-attr": ("5", "must", "dtd-styling"),
"class-on-p": ("4.1", "must", "dtd-attribute"),
"img-width": ("4.1", "must", "dtd-attribute"),
"img-no-alt": ("4.1", "must", "dtd-required-attribute"),
"input-type-email": ("4.1", "must", "dtd-value"),
"nested-table": ("4.2", "must", "dtd-nested-table"),
"table-in-li": ("4.2", "must", "dtd-nested-table"),
"bad-body-class": ("5.3", "must", "dtd-variant"),
"empty-body": ("4.1", "must", "dtd-content"),
"head-link-before-meta": ("3.3", "must", "dtd-head"),
"link-bad-rel": ("3.3", "must", "link-rel"),
}
@pytest.mark.parametrize(("name", "expected"), CASES.items())
def test_dtd_failure_maps_to_its_section(fail_page, name, expected):
report = validate_bytes(fail_page(name))
section, level, code = expected
match = [f for f in report.findings if f.code == code]
assert match, f"{name}: no {code} finding in {[f.code for f in report.findings]}"
assert match[0].section == section
assert match[0].level == level
assert not report.conforms
def test_one_mistake_is_one_finding(fail_page):
"""A <script> makes libxml2 say three things; the author sees one."""
report = validate_bytes(fail_page("script"))
assert [f.code for f in report.findings] == ["dtd-element"]
def test_findings_carry_a_line_number(fail_page):
report = validate_bytes(fail_page("div"))
assert report.findings[0].location.startswith("line ")
def test_messages_name_the_element(fail_page):
report = validate_bytes(fail_page("div"))
assert "<div>" in report.findings[0].message
def test_enumerated_values_come_from_the_dtd(fail_page):
report = validate_bytes(fail_page("input-type-email"))
message = report.findings[0].message
assert "text, password, checkbox, radio, submit, reset, hidden" in message
def test_unmapped_message_still_gets_a_section():
from mews.lint import _map_error
code, message = _map_error("Something libxml2 made up", set())
assert code == "dtd-other"
assert "Mews DTD" in message
def test_entities_the_doctype_defines_are_resolved(conforming):
report = validate_bytes(conforming(body="<p>a&nbsp;b&mdash;c</p>"))
assert report.conforms
def test_a_conforming_page_passes(conforming):
assert validate_bytes(conforming()).conforms

148
tests/test_lint_rules.py Normal file
View file

@ -0,0 +1,148 @@
"""The checks a DTD cannot express: prologue, head, sizes, images, transport."""
import pytest
from mews.fetch import Fetched
from mews.lint import SIZE_MUST, validate_bytes
CASES = {
"no-marker": ("3.2", "must", "marker"),
"two-markers": ("3.2", "must", "marker"),
"marker-wrong-version": ("3.2", "must", "marker-version"),
"meta-bad-name": ("3.3", "must", "meta-name"),
"two-stylesheets": ("3.3", "must", "link-stylesheet-count"),
"bad-feed-type": ("3.3", "must", "link-alternate-type"),
"no-xml-decl": ("3.1", "must", "xml-declaration"),
"bom": ("3.1", "must", "xml-declaration"),
"wrong-doctype": ("3.1", "must", "doctype"),
"internal-subset": ("3.1", "must", "internal-subset"),
"billion-laughs": ("3.1", "must", "internal-subset"),
"not-wellformed": ("3", "must", "well-formed"),
"wrong-namespace": ("3.1", "must", "root-element"),
"data-uri-img": ("4.2", "must", "image-data-uri"),
"no-lang": ("3.1", "should", "lang"),
"no-viewport": ("3.3", "should", "viewport"),
}
@pytest.mark.parametrize(("name", "expected"), CASES.items())
def test_rule_failure_maps_to_its_section(fail_page, name, expected):
report = validate_bytes(fail_page(name))
section, level, code = expected
match = [f for f in report.findings if f.code == code]
assert match, f"{name}: no {code} in {[f.code for f in report.findings]}"
assert (match[0].section, match[0].level) == (section, level)
def test_entity_declarations_are_refused_before_parsing(fail_page):
"""A page declaring its own entities is rejected without expanding them."""
report = validate_bytes(fail_page("billion-laughs"))
assert [f.code for f in report.findings] == ["internal-subset"]
def test_should_level_findings_still_conform(fail_page):
assert validate_bytes(fail_page("no-viewport")).conforms
assert validate_bytes(fail_page("no-lang")).conforms
def test_offsite_image_needs_the_page_address(fail_page):
"""Without a URL there is no site to compare against, so nothing is claimed."""
page = fail_page("offsite-img")
assert not [f for f in validate_bytes(page).findings if f.code == "image-offsite"]
report = validate_bytes(page, url="https://example.com/x.html")
assert [f.code for f in report.findings if f.code == "image-offsite"]
def test_subdomains_of_one_site_are_the_same_site(conforming):
page = conforming(body='<p><img src="https://img.example.com/a.png" alt="a" /></p>')
report = validate_bytes(page, url="https://example.com/x.html")
assert not [f for f in report.findings if f.code == "image-offsite"]
def test_oversized_page_is_reported_not_truncated(conforming):
filler = "<p>" + ("x" * 200) + "</p>\n "
page = conforming(body=filler * (SIZE_MUST // 200 + 20))
report = validate_bytes(page)
assert "size-must" in [f.code for f in report.findings]
assert not report.conforms
def test_page_between_the_two_size_limits_only_warns(conforming):
filler = "<p>" + ("x" * 200) + "</p>\n "
page = conforming(body=filler * 400)
report = validate_bytes(page)
assert "size-should" in [f.code for f in report.findings]
assert report.conforms
def _response(url="https://example.com/", headers=None, status=200):
import httpx
return Fetched(url, status, httpx.Headers(headers or {}), b"", False)
def test_a_cookie_on_the_page_is_a_failure(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(
headers={"set-cookie": "a=1", "etag": '"x"', "content-type": "text/html"}
),
)
cookie = [f for f in report.findings if f.code == "cookie"]
assert cookie and cookie[0].section == "7.4"
assert not report.conforms
def test_missing_cache_validators_warn(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(headers={"content-type": "text/html"}),
)
assert "validators" in [f.code for f in report.findings]
assert report.conforms
def test_plain_http_warns(conforming):
report = validate_bytes(
conforming(),
url="http://example.com/",
response=_response(
url="http://example.com/",
headers={"content-type": "text/html", "etag": '"x"'},
),
)
assert "https" in [f.code for f in report.findings]
def test_wrong_content_type_warns(conforming):
report = validate_bytes(
conforming(),
url="https://example.com/",
response=_response(headers={"content-type": "text/plain", "etag": '"x"'}),
)
assert "content-type" in [f.code for f in report.findings]
def test_title_and_description_are_picked_up(conforming):
report = validate_bytes(conforming())
assert report.title == "Test site"
assert report.description == "A test page."
assert report.language == "en"
def test_control_characters_are_stripped_from_the_title():
page = """<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"
"http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
<head>
<title>Good‮evil site</title>
<meta name="mews-profile" content="0.1" />
</head>
<body><p>x</p></body>
</html>
""".encode()
report = validate_bytes(page)
assert report.title == "Goodevil site"

65
tests/test_pages.py Normal file
View file

@ -0,0 +1,65 @@
"""The page emitter, against md2mews.py's.
mews.pages repeats md2mews.page() because md2mews.py is a standalone script
with its own dependencies. This is what keeps the two in step.
"""
import importlib.util
from pathlib import Path
import sys
import pytest
from mews import pages
ROOT = Path(__file__).resolve().parent.parent
def load_md2mews():
"""Import md2mews.py, which is a script rather than part of the package."""
spec = importlib.util.spec_from_file_location("md2mews", ROOT / "md2mews.py")
module = importlib.util.module_from_spec(spec)
sys.modules["md2mews"] = module
spec.loader.exec_module(module)
return module
CASES = [
("Plain", [" <p>x</p>"], "en", None, None, None),
("With a variant", [" <p>x</p>"], "en", None, "mews-cool", None),
("Right to left", [" <p>x</p>"], "he", "rtl", None, None),
("Described", [" <p>x</p>"], "en", None, "mews-warm", "A page."),
("An ampersand & a <tag>", [" <p>x</p>"], "en", None, None, None),
]
@pytest.mark.parametrize("case", CASES)
def test_the_two_emitters_agree(case):
"""Same prologue, head order, body attributes and escaping of & and <.
The two differ on quote characters in text, which md2mews.py escapes
because its one escape helper also fills attribute values. Both are
well-formed; mews.pages leaves them alone so prose reads as prose.
"""
md2mews = pytest.importorskip("mistune") and load_md2mews()
title, body, language, direction, variant, description = case
assert pages.page(title, body, language, direction, variant, description) == (
md2mews.page(title, body, language, direction, variant, description)
)
def test_a_message_page_conforms():
from mews.lint import validate_bytes
text = pages.message("Hello", ["One.", "Two."], [("/", "Home")])
assert validate_bytes(text.encode()).conforms
def test_quotes_in_text_are_left_alone():
text = pages.site_page('A "quoted" title', [" <p>x</p>"])
assert '<title>A "quoted" title</title>' in text
def test_an_apostrophe_is_not_escaped():
text = pages.message("Hello", ["It isn't escaped."], [("/", "Home")])
assert "isn't escaped" in text

166
tests/test_resources.py Normal file
View file

@ -0,0 +1,166 @@
"""The checks that read a page's stylesheet and images from a live server."""
from pathlib import Path
import zlib
import pytest
from mews.fetch import Fetcher
from mews.lint import validate_bytes
CANONICAL = (Path(__file__).resolve().parent.parent / "mews-0.1.css").read_bytes()
FONT_FACE = b"""@font-face {
font-family: "Atkinson Hyperlegible";
src: url(/fonts/atkinson.woff2) format("woff2");
}
"""
PNG = (
b"\x89PNG\r\n\x1a\n"
b"\x00\x00\x00\rIHDR\x00\x00\x00\x01\x00\x00\x00\x01\x08\x06\x00\x00\x00"
b"\x1f\x15\xc4\x89"
b"\x00\x00\x00\x00IEND\xaeB`\x82"
)
@pytest.fixture
def check(site):
"""Validate a page served by the local server, reading its sub-resources."""
def run(page: bytes, path: str = "/index.html"):
url = site.add(path, page, headers={"ETag": '"v1"'})
with Fetcher(allow_loopback=True) as fetcher:
return validate_bytes(page, url=url, fetcher=fetcher)
return run
def codes(report):
return [finding.code for finding in report.findings]
def test_an_unmodified_stylesheet_copy_passes(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
report = check(conforming())
assert "stylesheet-modified" not in codes(report)
def test_an_added_font_face_passes(site, check, conforming):
site.add("/mews-0.1.css", FONT_FACE + CANONICAL)
site.add("/fonts/atkinson.woff2", b"font")
report = check(conforming())
assert "stylesheet-modified" not in codes(report)
assert "font-offsite" not in codes(report)
def test_a_changed_stylesheet_fails(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL.replace(b"#faf8f3", b"#ffffff"))
report = check(conforming())
failure = [f for f in report.findings if f.code == "stylesheet-modified"]
assert failure and failure[0].section == "5.2"
assert not report.conforms
def test_a_font_from_another_site_fails(site, check, conforming):
off = FONT_FACE.replace(b"/fonts/atkinson.woff2", b"https://fonts.example.net/a")
site.add("/mews-0.1.css", off + CANONICAL)
report = check(conforming())
assert "font-offsite" in codes(report)
assert not report.conforms
def test_an_import_in_the_stylesheet_fails(site, check, conforming):
site.add("/mews-0.1.css", b'@import url("x.css");\n' + CANONICAL)
report = check(conforming())
assert "stylesheet-import" in codes(report)
def test_a_missing_stylesheet_only_warns(site, check, conforming):
report = check(conforming())
warning = [f for f in report.findings if f.code == "stylesheet-unreachable"]
assert warning and warning[0].level == "should"
def test_a_stylesheet_on_another_site_fails(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
report = check(conforming(stylesheet="https://cdn.example.net/mews-0.1.css"))
assert "stylesheet-offsite" in codes(report)
assert not report.conforms
def test_the_central_stylesheet_only_warns(site, check, conforming):
"""Linking mews.page's own copy is discouraged, not a failure."""
report = check(conforming(stylesheet="https://mews.page/mews-0.1.css"))
assert "stylesheet-offsite" not in codes(report)
canonical = [f for f in report.findings if f.code == "stylesheet-canonical"]
assert canonical and canonical[0].level == "should"
def test_a_cookie_on_the_stylesheet_fails(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL, headers={"Set-Cookie": "a=1"})
report = check(conforming())
assert "cookie" in codes(report)
assert not report.conforms
# --- images ------------------------------------------------------------
def test_a_small_png_passes(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
site.add("/a.png", PNG, headers={"Content-Type": "image/png"})
report = check(conforming(body='<p><img src="a.png" alt="A dot" /></p>'))
assert report.conforms
assert not report.warnings
def test_an_image_that_is_not_an_image_warns(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
site.add("/a.png", b"<svg/>")
report = check(conforming(body='<p><img src="a.png" alt="x" /></p>'))
warning = [f for f in report.findings if f.code == "image-format"]
assert warning and warning[0].level == "should"
assert report.conforms
def test_a_large_image_warns(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
site.add("/a.png", PNG + b"\x00" * 60_000)
report = check(conforming(body='<p><img src="a.png" alt="x" /></p>'))
assert "image-size" in codes(report)
def test_camera_metadata_warns(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
site.add("/a.jpg", b"\xff\xd8\xff\xe1\x00\x16Exif\x00\x00" + b"\x00" * 100)
report = check(conforming(body='<p><img src="a.jpg" alt="x" /></p>'))
assert "image-metadata" in codes(report)
def test_a_missing_image_warns(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
report = check(conforming(body='<p><img src="gone.png" alt="x" /></p>'))
assert "image-unreachable" in codes(report)
def test_only_the_first_twenty_images_are_read(site, check, conforming):
site.add("/mews-0.1.css", CANONICAL)
site.add("/a.png", PNG)
body = "".join(f'<p><img src="a.png?{i}" alt="x" /></p>' for i in range(25))
report = check(conforming(body=body))
assert "images-not-checked" in codes(report)
assert len([r for r in site.requests if r[0].startswith("/a.png")]) == 20
def test_a_compression_bomb_is_refused(site, conforming):
"""A short download that expands enormously is not a page."""
body = zlib.compress(b"x" * 10_000_000)
url = site.add(
"/big.html",
body,
headers={"Content-Encoding": "deflate", "Content-Type": "text/html"},
declare_length=False,
)
with Fetcher(allow_loopback=True) as fetcher:
response = fetcher.get(url)
assert response.truncated

365
tests/test_service.py Normal file
View file

@ -0,0 +1,365 @@
"""Submitting, listing, the endpoint, rechecks and the directory page."""
from datetime import timedelta
import functools
import pytest
from mews import app, check, db, pages
from mews.fetch import Fetcher, registered_domain
@pytest.fixture
def loopback(monkeypatch):
"""Let the service read the test server, and give it a registrable domain.
The test server answers on 127.0.0.1, which has no public suffix, so the
name it would be listed under is supplied here rather than relaxing the
rule in the service.
"""
monkeypatch.setattr(app, "Fetcher", functools.partial(Fetcher, allow_loopback=True))
monkeypatch.setattr(
check,
"registered_domain",
lambda host: "example.test" if host == "127.0.0.1" else registered_domain(host),
)
@pytest.fixture
def service(tmp_path, monkeypatch, loopback):
"""The app, wired to an empty database and a temporary directory page."""
database = tmp_path / "mews.db"
monkeypatch.setenv("MEWS_DB", str(database))
monkeypatch.setenv("MEWS_DIRECTORY", str(tmp_path / "directory.html"))
handle = db.connect(str(database))
db.init(handle)
handle.close()
return tmp_path
def post(url: str = "", *, listing: bool = False, client: str = "203.0.113.5"):
"""Send one form submission to the WSGI app and return status and body."""
from io import BytesIO
from urllib.parse import urlencode
fields = {"url": url, "list" if listing else "check": "x"}
data = urlencode(fields).encode()
captured: dict = {}
def start_response(status, headers):
captured["status"] = status
captured["headers"] = headers
body = b"".join(
app.application(
{
"PATH_INFO": "/result",
"REQUEST_METHOD": "POST",
"CONTENT_LENGTH": str(len(data)),
"wsgi.input": BytesIO(data),
"HTTP_X_REAL_IP": client,
},
start_response,
)
)
return captured, body.decode()
def test_a_conforming_page_is_listed(service, site, conforming):
url = site.add("/index.html", conforming())
captured, body = post(url, listing=True)
assert captured["status"] == "200 OK"
assert "Your site is listed" in body
handle = db.connect()
assert [row["domain"] for row in db.listed(handle)] == ["example.test"]
assert (service / "directory.html").exists()
assert "example.test" in (service / "directory.html").read_text()
def test_checking_does_not_list(service, site, conforming):
url = site.add("/index.html", conforming())
_, body = post(url, listing=False)
assert "This page conforms" in body
assert db.listed(db.connect()) == []
def test_a_failing_page_is_not_listed_and_says_why(service, site, fail_page):
url = site.add("/index.html", fail_page("div"))
_, body = post(url, listing=True)
assert "doesn't conform" in body
assert "Section 4.1" in body
assert db.listed(db.connect()) == []
def test_every_response_is_a_mews_page(service, site, conforming):
from mews.lint import validate_bytes
url = site.add("/index.html", conforming())
for listing in (False, True):
_, body = post(url, listing=listing)
assert validate_bytes(body.encode()).conforms
def test_no_response_sets_a_cookie(service, site, conforming):
url = site.add("/index.html", conforming())
captured, _ = post(url, listing=True)
assert not [name for name, _ in captured["headers"] if name.lower() == "set-cookie"]
def test_responses_are_not_cached(service, site, conforming):
url = site.add("/index.html", conforming())
captured, _ = post(url, listing=True)
assert ("Cache-Control", "no-store") in captured["headers"]
def test_an_address_that_is_not_a_page_is_refused(service):
_, body = post("http://10.0.0.1/", listing=True)
assert "domain name rather than an IP address" in body
def test_an_unreachable_page_is_reported(service, site):
_, body = post(site.base + "/missing.html", listing=True)
assert "it returned 404" in body
def test_resubmitting_a_site_updates_the_same_row(
service, site, conforming, monkeypatch
):
"""One row per registrable domain, however many pages of it are submitted."""
first = site.add("/a.html", conforming())
post(first, listing=True)
# Past the per-domain cooldown, which would otherwise turn this away.
later = db.now() + timedelta(hours=2)
monkeypatch.setattr(db, "now", lambda: later)
second = site.add("/b.html", conforming())
post(second, listing=True, client="203.0.113.6")
rows = db.listed(db.connect())
assert len(rows) == 1
assert rows[0]["url"] == second
def test_a_blocked_domain_is_refused(service, site, conforming):
handle = db.connect()
db.block(handle, "example.test", "spam")
url = site.add("/index.html", conforming())
_, body = post(url, listing=True)
assert "taken out of the directory" in body
def test_the_endpoint_answers_other_paths_and_methods_politely(service):
from io import BytesIO
captured: dict = {}
def start_response(status, headers):
captured["status"] = status
body = b"".join(
app.application(
{
"PATH_INFO": "/result",
"REQUEST_METHOD": "GET",
"wsgi.input": BytesIO(b""),
},
start_response,
)
).decode()
assert captured["status"].startswith("405")
assert "Nothing to show" in body
def test_rate_limits_stop_a_flood(connection):
client = db.ip_hash(connection, "203.0.113.9")
for _ in range(db.IP_LISTINGS_PER_HOUR):
assert (
db.rate_limited(connection, client=client, domain="a.example", listing=True)
is None
)
db.record_submission(
connection,
url="https://a.example/",
domain="other.example",
client=client,
outcome="listed",
)
assert "Try again in an hour" in db.rate_limited(
connection, client=client, domain="a.example", listing=True
)
def test_a_domain_has_to_wait_between_submissions(connection):
client = db.ip_hash(connection, "203.0.113.9")
db.record_submission(
connection,
url="https://a.example/",
domain="a.example",
client=client,
outcome="rejected",
)
assert "ten minutes" in db.rate_limited(
connection, client=client, domain="a.example", listing=True
)
def test_a_domain_that_keeps_failing_waits_a_day(connection, monkeypatch):
client = db.ip_hash(connection, "203.0.113.9")
base = db.now()
for index in range(db.DOMAIN_REJECTS_BEFORE_SLOWDOWN):
monkeypatch.setattr(db, "now", lambda i=index: base - timedelta(hours=i + 1))
db.record_submission(
connection,
url="https://a.example/",
domain="a.example",
client=client,
outcome="rejected",
)
monkeypatch.setattr(db, "now", lambda: base)
assert "try again tomorrow" in db.rate_limited(
connection, client=client, domain="a.example", listing=True
)
# --- rechecks ----------------------------------------------------------
def _list_site(connection, url):
return db.upsert_site(
connection,
domain="example.test",
url=url,
title="Test site",
description="",
language="en",
etag=None,
last_modified=None,
)[0]
def test_a_page_that_still_conforms_stays(connection, site, conforming):
url = site.add("/index.html", conforming(), headers={"ETag": '"v1"'})
site_id = _list_site(connection, url)
with Fetcher(allow_loopback=True) as fetcher:
assert (
check.recheck(
connection, db.site(connection, "example.test"), fetcher=fetcher
)
== "pass"
)
assert db.site(connection, "example.test")["state"] == "listed"
assert db.site(connection, "example.test")["etag"] == '"v1"'
assert site_id
def test_a_page_that_stopped_conforming_is_dropped(connection, site, fail_page):
url = site.add("/index.html", fail_page("div"))
_list_site(connection, url)
with Fetcher(allow_loopback=True) as fetcher:
assert (
check.recheck(
connection, db.site(connection, "example.test"), fetcher=fetcher
)
== "fail"
)
row = db.site(connection, "example.test")
assert row["state"] == "removed"
assert "Section 4.1" in row["reason"]
assert db.listed(connection) == []
def test_a_not_modified_answer_passes_without_rereading(connection, site, conforming):
url = site.add("/index.html", conforming(), headers={"ETag": '"v1"'})
_list_site(connection, url)
site.add("/index.html", b"", status=304, headers={"ETag": '"v1"'})
connection.execute("UPDATE sites SET etag = '\"v1\"'")
with Fetcher(allow_loopback=True) as fetcher:
assert (
check.recheck(
connection, db.site(connection, "example.test"), fetcher=fetcher
)
== "pass"
)
assert db.site(connection, "example.test")["state"] == "listed"
assert "If-None-Match" in site.requests[-1][1]
def test_downtime_takes_several_tries_to_drop_a_site(connection, site, conforming):
url = site.add("/index.html", conforming())
_list_site(connection, url)
site.routes.clear()
site.add("/index.html", b"broken", status=503)
for _ in range(db.TRANSIENT_LIMIT - 1):
with Fetcher(allow_loopback=True) as fetcher:
assert (
check.recheck(
connection, db.site(connection, "example.test"), fetcher=fetcher
)
== "unreachable"
)
assert db.site(connection, "example.test")["state"] == "listed"
with Fetcher(allow_loopback=True) as fetcher:
assert (
check.recheck(
connection, db.site(connection, "example.test"), fetcher=fetcher
)
== "fail"
)
assert db.site(connection, "example.test")["state"] == "removed"
def test_only_sites_that_are_due_are_rechecked(connection, monkeypatch):
_list_site(connection, "https://example.test/")
assert db.due(connection) == []
later = db.now() + timedelta(days=db.RECHECK_DAYS + 1)
monkeypatch.setattr(db, "now", lambda: later)
assert len(db.due(connection)) == 1
# --- the directory page ------------------------------------------------
def test_the_directory_page_conforms(connection, tmp_path):
from mews.lint import validate_bytes
_list_site(connection, "https://example.test/")
target = tmp_path / "directory.html"
pages.write_directory(str(target), db.listed(connection))
assert validate_bytes(target.read_bytes()).conforms
def test_a_hostile_title_is_escaped_and_stripped(connection, tmp_path):
from mews.lint import validate_bytes
connection.execute(
"INSERT INTO sites (domain, url, title, state, listed_at) "
"VALUES ('x.test', 'https://x.test/', '<script>‮oops</script>', "
"'listed', '2026-01-01T00:00:00+00:00')"
)
connection.commit()
target = tmp_path / "directory.html"
pages.write_directory(str(target), db.listed(connection))
text = target.read_text()
assert "<script>" not in text
assert "&lt;script&gt;" in text
assert validate_bytes(target.read_bytes()).conforms
def test_an_empty_directory_still_conforms(connection, tmp_path):
from mews.lint import validate_bytes
target = tmp_path / "directory.html"
pages.write_directory(str(target), db.listed(connection))
assert validate_bytes(target.read_bytes()).conforms
assert "Nothing here yet" in target.read_text()
def test_a_page_that_would_not_conform_leaves_the_old_one_alone(
connection, tmp_path, monkeypatch
):
target = tmp_path / "directory.html"
pages.write_directory(str(target), db.listed(connection))
before = target.read_text()
monkeypatch.setattr(pages, "directory", lambda rows: "<p>not a page</p>")
with pytest.raises(ValueError, match="doesn't conform"):
pages.write_directory(str(target), db.listed(connection))
assert target.read_text() == before

35
tests/test_site.py Normal file
View file

@ -0,0 +1,35 @@
"""The project's own pages. Every one of them has to follow the spec."""
from pathlib import Path
import subprocess
import sys
import pytest
from mews.lint import validate_file
ROOT = Path(__file__).resolve().parent.parent
PAGES = [*sorted((ROOT / "site").glob("*.html")), ROOT / "sample" / "sample.html"]
@pytest.mark.parametrize("page", PAGES, ids=lambda p: p.name)
def test_the_projects_own_pages_conform(page):
report = validate_file(str(page))
assert report.conforms, "\n".join(str(f) for f in report.failures)
@pytest.mark.parametrize("page", PAGES, ids=lambda p: p.name)
def test_the_projects_own_pages_have_no_warnings_either(page):
report = validate_file(str(page))
assert not report.warnings, "\n".join(str(f) for f in report.warnings)
def test_the_dtd_still_matches_the_spec():
"""dtd/check_dtd.py is hand-maintained; keep it running in CI."""
result = subprocess.run(
[sys.executable, str(ROOT / "dtd" / "check_dtd.py")],
capture_output=True,
text=True,
check=False,
)
assert result.returncode == 0, result.stderr