diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..2e5cf78 --- /dev/null +++ b/.gitignore @@ -0,0 +1,10 @@ +build/ +result +__pycache__/ +.venv/ +.ruff_cache/ +.pytest_cache/ +*.db +*.db-wal +*.db-shm +directory.html diff --git a/flake.lock b/flake.lock new file mode 100644 index 0000000..0411be3 --- /dev/null +++ b/flake.lock @@ -0,0 +1,99 @@ +{ + "nodes": { + "nixpkgs": { + "locked": { + "lastModified": 1791594108, + "narHash": "sha256-XK4KceonkwD8LuMAt/vNrz982mXJM5jeHAxMwsUXXAQ=", + "owner": "nixos", + "repo": "nixpkgs", + "rev": "6510408d147d8c5e6d76043e899e9251756f6a75", + "type": "github" + }, + "original": { + "owner": "nixos", + "ref": "nixos-unstable", + "repo": "nixpkgs", + "type": "github" + } + }, + "pyproject-build-systems": { + "inputs": { + "nixpkgs": [ + "nixpkgs" + ], + "pyproject-nix": [ + "pyproject-nix" + ], + "uv2nix": [ + "uv2nix" + ] + }, + "locked": { + "lastModified": 1791170410, + "narHash": "sha256-pyru9MpuXRiyq0Bd4q0nHpL4T6HCuZT4B+alcgAYE4A=", + "owner": "pyproject-nix", + "repo": "build-system-pkgs", + "rev": "953d35d03a51dbecaf3feb3fb679eb63e90ebeb4", + "type": "github" + }, + "original": { + "owner": "pyproject-nix", + "repo": "build-system-pkgs", + "type": "github" + } + }, + "pyproject-nix": { + "inputs": { + "nixpkgs": [ + "nixpkgs" + ] + }, + "locked": { + "lastModified": 1791657700, + "narHash": "sha256-vcCAvhtA7hG+/gOge29hYHy7I/dLkIsr5nLx4sFEQic=", + "owner": "pyproject-nix", + "repo": "pyproject.nix", + "rev": "ddc400a6883b042646d0940e65971bb7f7e96107", + "type": "github" + }, + "original": { + "owner": "pyproject-nix", + "repo": "pyproject.nix", + "type": "github" + } + }, + "root": { + "inputs": { + "nixpkgs": "nixpkgs", + "pyproject-build-systems": "pyproject-build-systems", + "pyproject-nix": "pyproject-nix", + "uv2nix": "uv2nix" + } + }, + "uv2nix": { + "inputs": { + "nixpkgs": [ + "nixpkgs" + ], + "pyproject-nix": [ + "pyproject-nix" + ] + }, + "locked": { + "lastModified": 1791656763, + "narHash": "sha256-2mFllBdFGB4zNogzx02b8O5JObd7LeAGsndk+ZB5XvI=", + "owner": "pyproject-nix", + "repo": "uv2nix", + "rev": "ddca81d752e29c063eaa914e040c3ec56ed39558", + "type": "github" + }, + "original": { + "owner": "pyproject-nix", + "repo": "uv2nix", + "type": "github" + } + } + }, + "root": "root", + "version": 7 +} diff --git a/flake.nix b/flake.nix new file mode 100644 index 0000000..9928aab --- /dev/null +++ b/flake.nix @@ -0,0 +1,129 @@ +{ + description = "Mews Profile: page validator, site directory and the mews.page site"; + + inputs = { + nixpkgs.url = "github:nixos/nixpkgs/nixos-unstable"; + + pyproject-nix = { + url = "github:pyproject-nix/pyproject.nix"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + + uv2nix = { + url = "github:pyproject-nix/uv2nix"; + inputs.pyproject-nix.follows = "pyproject-nix"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + + pyproject-build-systems = { + url = "github:pyproject-nix/build-system-pkgs"; + inputs.pyproject-nix.follows = "pyproject-nix"; + inputs.uv2nix.follows = "uv2nix"; + inputs.nixpkgs.follows = "nixpkgs"; + }; + }; + + outputs = + { + nixpkgs, + pyproject-nix, + uv2nix, + pyproject-build-systems, + ... + }: + let + inherit (nixpkgs) lib; + forAllSystems = lib.genAttrs lib.systems.flakeExposed; + + workspace = uv2nix.lib.workspace.loadWorkspace { workspaceRoot = ./.; }; + + # Wheels, not sources: the service is built on the machine it runs on, + # which has one CPU, and lxml from source there takes minutes. + overlay = workspace.mkPyprojectOverlay { + sourcePreference = "wheel"; + }; + + editableOverlay = workspace.mkEditablePyprojectOverlay { + root = "$REPO_ROOT"; + }; + + pythonSets = forAllSystems ( + system: + let + pkgs = nixpkgs.legacyPackages.${system}; + in + (pkgs.callPackage pyproject-nix.build.packages { + python = pkgs.python312; + }).overrideScope + ( + lib.composeManyExtensions [ + pyproject-build-systems.overlays.wheel + overlay + ] + ) + ); + + specDescription = "The Mews Profile specification, version 0.1."; + in + { + devShells = forAllSystems ( + system: + let + pkgs = nixpkgs.legacyPackages.${system}; + pythonSet = pythonSets.${system}.overrideScope editableOverlay; + virtualenv = pythonSet.mkVirtualEnv "mews-dev-env" workspace.deps.all; + in + { + default = pkgs.mkShell { + packages = [ + virtualenv + pkgs.uv + # md2mews.py is a standalone script with its own dependency + # header, so it is not part of the workspace environment. + (pkgs.python3.withPackages (ps: [ ps.mistune ])) + ]; + env = { + UV_NO_SYNC = "1"; + UV_PYTHON = pythonSet.python.interpreter; + UV_PYTHON_DOWNLOADS = "never"; + }; + shellHook = '' + unset PYTHONPATH + export REPO_ROOT=$(git rev-parse --show-toplevel) + ''; + }; + } + ); + + packages = forAllSystems ( + system: + let + pkgs = nixpkgs.legacyPackages.${system}; + app = pythonSets.${system}.mkVirtualEnv "mews-env" workspace.deps.default; + converter = pkgs.python3.withPackages (ps: [ ps.mistune ]); + in + { + default = app; + + # The published site. Every page it assembles is validated here, so a + # page that does not follow the spec cannot reach the server. + site = pkgs.runCommand "mews-site" { } '' + mkdir -p $out/spec $out/dtd + cp ${./site}/index.html ${./site}/about.html ${./site}/check.html $out/ + cp ${./mews-0.1.css} $out/mews-0.1.css + cp ${./mews-0.1.css} $out/spec/mews-0.1.css + cp ${./dtd/mews-0.1.dtd} $out/dtd/mews-0.1.dtd + cp ${./dtd/xhtml-mobile12.dtd} $out/dtd/xhtml-mobile12.dtd + chmod -R u+w $out + + ${converter}/bin/python ${./md2mews.py} ${./doc/SPEC.md} \ + -o $out/spec/0.1.html --lang en --variant mews-cool \ + --description ${lib.escapeShellArg specDescription} + + ${app}/bin/mewslint --quiet \ + $out/index.html $out/about.html $out/check.html $out/spec/0.1.html + ''; + } + ); + }; +} diff --git a/mews/__init__.py b/mews/__init__.py new file mode 100644 index 0000000..2ae0164 --- /dev/null +++ b/mews/__init__.py @@ -0,0 +1,6 @@ +"""Page validator and site directory for the Mews Profile. + +See doc/SPEC.md for the format this package checks. +""" + +VERSION = "0.1" diff --git a/mews/css.py b/mews/css.py new file mode 100644 index 0000000..0332119 --- /dev/null +++ b/mews/css.py @@ -0,0 +1,136 @@ +"""Comparing a site's stylesheet against the default one. + +SPEC.md 5.2 lets a site host its own copy of the default stylesheet and change +nothing in it, except to add @font-face rules that load fonts from the site +itself. Checking that means comparing text, since a site may legitimately +differ in line endings and trailing whitespace after a copy-paste. +""" + +from dataclasses import dataclass, field +import difflib +from functools import cache +from importlib import resources +import re + +FONT_FACE = re.compile(r"@font-face\b", re.IGNORECASE) +URL_IN_SRC = re.compile(r"url\(\s*['\"]?([^'\")]+)", re.IGNORECASE) +IMPORT_RULE = re.compile(r"@import\b", re.IGNORECASE) + + +@dataclass +class Comparison: + """What a candidate stylesheet differs from the default by.""" + + equal: bool + diff_line: int | None = None + diff: str = "" + font_faces: list[str] = field(default_factory=list) + has_import: bool = False + + @property + def font_urls(self) -> list[str]: + """Every url() the added @font-face rules load.""" + return [ + match.group(1).strip() + for block in self.font_faces + for match in URL_IN_SRC.finditer(block) + ] + + +@cache +def canonical() -> str: + """Return the normalised default stylesheet shipped with this package.""" + data = resources.files("mews.data").joinpath("mews-0.1.css").read_text("utf-8") + return normalise(data) + + +def normalise(text: str) -> str: + """Strip the differences a copy may pick up without changing any rule.""" + text = text.lstrip("").replace("\r\n", "\n").replace("\r", "\n") + lines = [line.rstrip() for line in text.split("\n")] + out: list[str] = [] + for line in lines: + if line or (out and out[-1]): + out.append(line) + return "\n".join(out).strip("\n") + + +def strip_font_faces(text: str) -> tuple[str, list[str]]: + """Remove top-level @font-face blocks, returning the rest and the blocks. + + Only brace depth zero counts, so an @font-face inside a media query is left + in place and shows up as a difference. + """ + blocks: list[str] = [] + out: list[str] = [] + index = 0 + depth = 0 + while index < len(text): + if depth == 0 and FONT_FACE.match(text, index): + end = _block_end(text, index) + if end is None: + break + blocks.append(text[index:end]) + index = end + continue + character = text[index] + if character == "{": + depth += 1 + elif character == "}": + depth = max(0, depth - 1) + out.append(character) + index += 1 + out.append(text[index:]) + return "".join(out), blocks + + +def _block_end(text: str, start: int) -> int | None: + """Index just past the brace-balanced block beginning at start.""" + opened = text.find("{", start) + if opened == -1: + return None + depth = 0 + for index in range(opened, len(text)): + if text[index] == "{": + depth += 1 + elif text[index] == "}": + depth -= 1 + if depth == 0: + return index + 1 + return None + + +def compare(candidate: str) -> Comparison: + """Compare a stylesheet against the default, allowing added @font-face rules.""" + expected = canonical() + if normalise(candidate) == expected: + return Comparison(equal=True) + + remainder, blocks = strip_font_faces(candidate) + found = normalise(remainder) + has_import = bool(IMPORT_RULE.search(candidate)) + if found == expected: + return Comparison(equal=True, font_faces=blocks, has_import=has_import) + + diff = list( + difflib.unified_diff( + expected.split("\n"), found.split("\n"), "mews-0.1.css", "your copy", n=1 + ) + ) + line = next( + ( + index + 1 + for index, (left, right) in enumerate( + zip(expected.split("\n"), found.split("\n"), strict=False) + ) + if left != right + ), + min(len(expected.split("\n")), len(found.split("\n"))) + 1, + ) + return Comparison( + equal=False, + diff_line=line, + diff="\n".join(diff[:8]), + font_faces=blocks, + has_import=has_import, + ) diff --git a/mews/data/README.md b/mews/data/README.md new file mode 100644 index 0000000..198f790 --- /dev/null +++ b/mews/data/README.md @@ -0,0 +1,7 @@ +# Packaged copies + +The canonical files live at the repository root (`mews-0.1.css`) and in `dtd/` +(`mews-0.1.dtd`), where SPEC.md and `dtd/check_dtd.py` reference them. These +copies exist so the validator still finds them when it is installed as a wheel, +with nothing but the package on disk. `tests/test_data.py` fails if a copy and +its original drift apart. diff --git a/mews/data/mews-0.1.css b/mews/data/mews-0.1.css new file mode 100644 index 0000000..fc0d8ed --- /dev/null +++ b/mews/data/mews-0.1.css @@ -0,0 +1,106 @@ +/* Mews Profile default stylesheet 0.1 */ + +/* Base: light theme */ +body { + background-color: #faf8f3; + color: #1f1f1f; + font-family: "Atkinson Hyperlegible", Verdana, Tahoma, system-ui, sans-serif; + font-size: 106%; + line-height: 1.5; + margin: 0 auto; + padding: 1em; + max-width: 40em; +} + +h1, h2, h3, h4, h5, h6 { + font-weight: bold; + line-height: 1.25; + margin: 1.6em 0 0.5em 0; +} +h1 { font-size: 1.6em; margin-top: 0.5em; } +h2 { font-size: 1.3em; } +h3 { font-size: 1.1em; } +h4, h5, h6 { font-size: 1em; } + +p, ul, ol, dl, blockquote, pre, table, address { + margin: 0 0 1em 0; +} +ul, ol { padding-left: 1.5em; } +li { margin-bottom: 0.25em; } +dt { font-weight: bold; } +dd { margin: 0 0 0.5em 1.5em; } + +a { color: #1a4f9c; text-decoration: underline; } +a:visited { color: #5b2a86; } +a:focus, a:active { outline: 2px solid #1a4f9c; } + +blockquote { + margin-left: 0; + padding-left: 1em; + border-left: 3px solid #c9c4b8; +} + +pre, code, kbd, samp { + font-family: "Source Code Pro", Consolas, Menlo, "DejaVu Sans Mono", monospace; + font-size: 0.9em; +} +pre { + background-color: #efece4; + padding: 0.75em; + overflow: auto; +} + +hr { + border: 0; + border-top: 1px solid #c9c4b8; + margin: 2em 0; +} + +table { border-collapse: collapse; } +th, td { + border: 1px solid #c9c4b8; + padding: 0.3em 0.6em; + text-align: left; + vertical-align: top; +} +caption { font-weight: bold; text-align: left; } + +img { + display: block; + max-width: 100%; + height: auto; + border: 0; +} + +input, select, textarea { font-size: 1em; font-family: inherit; } + +/* Body variants: light */ +body.mews-warm { background-color: #fbf4e8; color: #2a2420; } +body.mews-warm a { color: #8a3d14; } +body.mews-cool { background-color: #f3f6fa; color: #1c2430; } +body.mews-cool a { color: #1a4f9c; } +body.mews-green { background-color: #f2f6ef; color: #1f2a1f; } +body.mews-green a { color: #24502a; } +body.mews-mono { background-color: #e8ebe1; color: #1e2219; } +body.mews-mono a { color: #1e2219; } + +/* Dark theme: follows the system setting */ +@media (prefers-color-scheme: dark) { + body { background-color: #18191b; color: #e3e1dc; } + a { color: #8fb8f5; } + a:visited { color: #c9a3f2; } + a:focus, a:active { outline-color: #8fb8f5; } + blockquote, hr, th, td { border-color: #45464a; } + pre { background-color: #222326; } + + body.mews-warm { background-color: #1d1916; color: #e8dfd3; } + body.mews-warm a { color: #f0b48a; } + body.mews-cool { background-color: #161a20; color: #dde4ee; } + body.mews-cool a { color: #8fb8f5; } + body.mews-green { background-color: #161b16; color: #dbe6d8; } + body.mews-green a { color: #9fd39a; } + body.mews-mono { background-color: #1b1d18; color: #c8d0b8; } + body.mews-mono a { color: #c8d0b8; } +} + +html { color-scheme: light dark; } \ No newline at end of file diff --git a/mews/data/mews-0.1.dtd b/mews/data/mews-0.1.dtd new file mode 100644 index 0000000..e09c5da --- /dev/null +++ b/mews/data/mews-0.1.dtd @@ -0,0 +1,396 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/mews/fetch.py b/mews/fetch.py new file mode 100644 index 0000000..bf94dc8 --- /dev/null +++ b/mews/fetch.py @@ -0,0 +1,382 @@ +"""Fetching pages that strangers ask the checker to look at. + +Every URL reaching this module is author-supplied, so the service must not be +usable as a probe against the machine it runs on or its neighbours. Addresses +are vetted before the connection and the connection is pinned to the vetted +address, so a second DNS answer cannot redirect it (DNS rebinding). +""" + +from dataclasses import dataclass, replace +import ipaddress +import os +import re +import socket +import time +from urllib.parse import urlsplit, urlunsplit + +import httpx +from publicsuffixlist import PublicSuffixList + +# Read caps, in bytes. The page cap is above the 256 KB that SPEC.md 4.3 makes a +# MUST so an oversized page is reported as too large rather than unreachable. +PAGE_CAP = 288 * 1024 +IMAGE_CAP = 64 * 1024 +CSS_CAP = 256 * 1024 + +MAX_REDIRECTS = 3 +MAX_URL_LENGTH = 2048 +USER_AGENT = "mews.page checker (+https://mews.page/about)" + +_PSL = PublicSuffixList() + +# Ranges ipaddress does not classify but which must never be reached: carrier +# NAT (which is also Tailscale's range), IETF protocol assignment, benchmarking, +# the documentation ranges, and the IPv6 transition mechanisms that tunnel an +# IPv4 address inside an address that looks global. +_BLOCKED_NETS = [ + ipaddress.ip_network(net) + for net in ( + "100.64.0.0/10", + "192.0.0.0/24", + "198.18.0.0/15", + "192.0.2.0/24", + "198.51.100.0/24", + "203.0.113.0/24", + "240.0.0.0/4", + "255.255.255.255/32", + "2001:db8::/32", + "2002::/16", + "2001::/32", + "64:ff9b::/96", + ) +] + +# Hosts whose own addresses must be unreachable, as CIDRs. The deployment sets +# this to the machine's public address so a submission cannot be aimed at a +# service sharing the box. +_EXTRA_NETS = [ + ipaddress.ip_network(net.strip()) + for net in os.environ.get("MEWS_BLOCK_NETS", "").split(",") + if net.strip() +] + +_LOCAL_SUFFIXES = (".local", ".internal", ".home.arpa", ".localhost") + +# What a host name may be made of. Checking this first means a typo gets a +# clearer answer than a complaint about public suffixes. +_HOSTNAME = re.compile( + r"^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)*$" +) + + +class UrlError(Exception): + """A URL the checker will not fetch. The message is shown to the author.""" + + +class FetchError(Exception): + """A fetch that produced no page. The message is shown to the author.""" + + +@dataclass(frozen=True) +class Fetched: + """One response the checker read, with the body it kept.""" + + url: str + status: int + headers: httpx.Headers + body: bytes + truncated: bool + scheme_downgraded: bool = False + + +def registered_domain(host: str) -> str | None: + """Return the registrable domain of host, or None when it has no public suffix. + + Comparing registrable domains rather than hostnames is what makes the + same-site rules in SPEC.md 4.2 and 7.4 mean anything: a free subdomain host + would otherwise let any two unrelated sites count as one. + """ + if not host: + return None + try: + name = host.strip().rstrip(".").lower().encode("idna").decode("ascii") + except (UnicodeError, UnicodeDecodeError): + return None + return _PSL.privatesuffix(name) + + +def same_site(a: str, b: str) -> bool: + """Whether two hosts share a registrable domain.""" + first = registered_domain(a) + return first is not None and first == registered_domain(b) + + +def normalise_url(raw: str, *, allow_loopback: bool = False) -> str: + """Return a URL the checker is willing to fetch, or raise UrlError. + + This runs before any name lookup, so a rejected URL costs nothing. With + allow_loopback the rules relax enough to reach a test server on this + machine: an address literal, localhost, and any port. + """ + text = (raw or "").strip() + if not text: + raise UrlError("Enter the address of a page to check.") + if len(text) > MAX_URL_LENGTH: + raise UrlError("That address is too long.") + if "://" not in text: + text = "https://" + text + + parts = urlsplit(text) + if parts.scheme not in ("http", "https"): + raise UrlError("Enter an address that starts with http:// or https://.") + if "@" in parts.netloc: + raise UrlError("Enter an address without a user name in it.") + + try: + host, port = parts.hostname, parts.port + except ValueError as error: + raise UrlError("That address has a port the checker can't read.") from error + if not host: + raise UrlError( + "That doesn't look like a web address. Enter the full address " + "of a page, like https://example.com/" + ) + if port is not None and port not in (80, 443) and not allow_loopback: + raise UrlError("The checker reads pages on the usual web ports only.") + + host = host.lower().rstrip(".") + if _is_ip_literal(host): + if not (allow_loopback and _is_loopback_literal(host)): + raise UrlError("Enter a domain name rather than an IP address.") + elif host == "localhost" or host.endswith(_LOCAL_SUFFIXES): + if not allow_loopback: + raise UrlError("That address is only reachable on a local network.") + elif not _HOSTNAME.match(_ascii(host)): + raise UrlError( + "That doesn't look like a web address. Enter the full address " + "of a page, like https://example.com/" + ) + elif registered_domain(host) is None: + raise UrlError("That domain name isn't one the checker can reach.") + + netloc = host if port is None else f"{host}:{port}" + return urlunsplit((parts.scheme, netloc, parts.path or "/", parts.query, "")) + + +def _ascii(host: str) -> str: + """Return the punycode form of a host name, or the name unchanged.""" + try: + return host.encode("idna").decode("ascii") + except (UnicodeError, UnicodeDecodeError): + return host + + +def _is_ip_literal(host: str) -> bool: + """Whether host is an address literal in any of the forms a parser accepts.""" + candidate = host.strip("[]") + try: + ipaddress.ip_address(candidate) + except ValueError: + pass + else: + return True + # Decimal, octal and hex integer forms of an IPv4 address, which urlsplit + # leaves alone but a resolver would accept. + if host.isdigit(): + return True + bare = host.replace(".", "") + return bare.startswith(("0x", "0X")) or ( + host.startswith("0") and len(host) > 1 and bare.isdigit() + ) + + +def _is_loopback_literal(host: str) -> bool: + """Whether host is a literal address on this machine.""" + try: + return ipaddress.ip_address(host.strip("[]")).is_loopback + except ValueError: + return False + + +def _embedded(address: ipaddress.IPv6Address) -> ipaddress.IPv4Address | None: + """Return the IPv4 address a transition mechanism hides inside an IPv6 one.""" + for attribute in ("ipv4_mapped", "sixtofour"): + value = getattr(address, attribute, None) + if value is not None: + return value + teredo = getattr(address, "teredo", None) + if teredo: + return teredo[1] + if address in ipaddress.ip_network("64:ff9b::/96"): + return ipaddress.IPv4Address(int(address) & 0xFFFFFFFF) + return None + + +def vet_address(address: str, *, allow_loopback: bool = False) -> None: + """Raise UrlError unless address is a public one the checker may connect to.""" + ip = ipaddress.ip_address(address) + if allow_loopback and ip.is_loopback: + return + if ( + ip.is_private + or ip.is_loopback + or ip.is_link_local + or ip.is_reserved + or ip.is_multicast + or ip.is_unspecified + or any(ip in net for net in _BLOCKED_NETS) + or any(ip in net for net in _EXTRA_NETS) + ): + raise UrlError("That address isn't on the public internet.") + if isinstance(ip, ipaddress.IPv6Address): + inner = _embedded(ip) + if inner is not None: + vet_address(str(inner), allow_loopback=allow_loopback) + + +def resolve(url: str, *, allow_loopback: bool = False) -> str: + """Return one vetted address for the URL's host, rejecting the host if any fails. + + Every answer has to pass: a host that resolves to one public and one private + address would otherwise be a coin toss. + """ + parts = urlsplit(url) + host = parts.hostname or "" + port = parts.port or (443 if parts.scheme == "https" else 80) + try: + infos = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM) + except socket.gaierror as error: + raise FetchError( + "We couldn't find that domain name. Check the address and try again." + ) from error + addresses = [info[4][0] for info in infos] + if not addresses: + raise FetchError("We couldn't find that domain name.") + for address in addresses: + vet_address(address, allow_loopback=allow_loopback) + return addresses[0] + + +class Fetcher: + """Reads pages and their sub-resources, under a time and byte budget. + + One Fetcher serves one check, so the whole-check budget is shared across the + page, its stylesheet and its images. + """ + + def __init__(self, *, allow_loopback: bool = False, budget: float = 45.0): + self.allow_loopback = allow_loopback + self._deadline = time.monotonic() + budget + self._client = httpx.Client( + follow_redirects=False, + trust_env=False, + http2=False, + verify=True, + timeout=httpx.Timeout(connect=5.0, read=10.0, write=5.0, pool=5.0), + limits=httpx.Limits(max_connections=4, max_keepalive_connections=2), + headers={ + "User-Agent": USER_AGENT, + "Accept": "text/html,application/xhtml+xml", + "Accept-Encoding": "gzip", + }, + ) + + def __enter__(self) -> "Fetcher": + return self + + def __exit__(self, *exc: object) -> None: + self.close() + + def close(self) -> None: + """Close the underlying connections.""" + self._client.close() + + def get( + self, + url: str, + *, + cap: int = PAGE_CAP, + headers: dict[str, str] | None = None, + ) -> Fetched: + """Fetch one URL, following redirects by hand and vetting each hop.""" + current = normalise_url(url, allow_loopback=self.allow_loopback) + downgraded = False + for _ in range(MAX_REDIRECTS + 1): + response = self._request(current, cap=cap, headers=headers) + if response.status not in (301, 302, 303, 307, 308): + if downgraded: + return replace(response, scheme_downgraded=True) + return response + location = response.headers.get("location", "") + if not location: + raise FetchError("That page redirects without saying where to.") + target = normalise_url( + str(httpx.URL(current).join(location)), + allow_loopback=self.allow_loopback, + ) + was_secure = urlsplit(current).scheme == "https" + if was_secure and urlsplit(target).scheme == "http": + downgraded = True + current = target + raise FetchError("That page redirects too many times.") + + def _request( + self, + url: str, + *, + cap: int, + headers: dict[str, str] | None, + ) -> Fetched: + """Make one pinned request, streaming the body up to cap bytes.""" + self._check_deadline() + parts = urlsplit(url) + host = parts.hostname or "" + port = parts.port + address = resolve(url, allow_loopback=self.allow_loopback) + literal = f"[{address}]" if ":" in address else address + pinned = urlunsplit( + ( + parts.scheme, + literal if port is None else f"{literal}:{port}", + parts.path, + parts.query, + "", + ) + ) + request_headers = {"Host": parts.netloc, **(headers or {})} + extensions = {"sni_hostname": host} if parts.scheme == "https" else {} + try: + with self._client.stream( + "GET", pinned, headers=request_headers, extensions=extensions + ) as response: + declared = response.headers.get("content-length") + if declared and declared.isdigit() and int(declared) > cap: + raise FetchError("That page is too large for the checker to read.") + body, truncated = self._read(response, cap) + except httpx.TimeoutException as error: + raise FetchError("That page took too long to answer.") from error + except httpx.HTTPError as error: + raise FetchError("We couldn't connect to that site.") from error + return Fetched(url, response.status_code, response.headers, body, truncated) + + def _read(self, response: httpx.Response, cap: int) -> tuple[bytes, bool]: + """Read a streamed body, stopping at cap decoded bytes.""" + chunks: list[bytes] = [] + total = 0 + for chunk in response.iter_bytes(): + self._check_deadline() + chunks.append(chunk) + total += len(chunk) + if total > cap: + return b"".join(chunks)[:cap], True + body = b"".join(chunks) + encoded = response.num_bytes_downloaded or len(body) + # A body that expanded enormously from a small download is a compression + # bomb, not a page. + if encoded and len(body) > 100 * encoded and len(body) > 64 * 1024: + raise FetchError("That page is compressed in a way the checker won't read.") + return body, False + + def _check_deadline(self) -> None: + if time.monotonic() > self._deadline: + raise FetchError("Checking that page took too long.") diff --git a/mews/lint.py b/mews/lint.py new file mode 100644 index 0000000..3db2b44 --- /dev/null +++ b/mews/lint.py @@ -0,0 +1,976 @@ +"""The Mews page validator. + +SPEC.md 9.1 defines conformance as two checks: the page validates against +dtd/mews-0.1.dtd, and it follows the rules a DTD cannot express. This module +runs both and reports every failure with its spec section. + +Failures at MUST level mean the page does not conform. Failures at SHOULD level +are warnings: they are worth fixing but do not make a page non-conforming. +""" + +import argparse +from dataclasses import dataclass, field +import html.entities +from importlib import resources +from pathlib import Path +import re +import sys +from urllib.parse import urljoin, urlsplit + +from lxml import etree + +from mews import VERSION, css as mews_css +from mews.fetch import ( + CSS_CAP, + IMAGE_CAP, + Fetched, + Fetcher, + FetchError, + UrlError, + normalise_url, + same_site, +) + +XHTML = "http://www.w3.org/1999/xhtml" +XML_LANG = "{http://www.w3.org/XML/1998/namespace}lang" + +MARKER = "mews-profile" +CANONICAL_STYLESHEET = "https://mews.page/mews-0.1.css" + +OK = 200 +MAX_FINDINGS = 50 +MAX_IMAGES = 20 + +SIZE_MUST = 256 * 1024 +SIZE_SHOULD = 64 * 1024 +SIZE_TOTAL = 320 * 1024 +IMAGE_SHOULD = 50 * 1024 + +XML_DECLARATION = re.compile(rb'^<\?xml version="1\.0" encoding="(?i:UTF-8)"\?>') +DOCTYPE = re.compile( + rb'' +) + +ALLOWED_META = ("mews-profile", "description", "author", "viewport") +ALLOWED_RELS = ("stylesheet", "icon", "alternate") + +# Magic bytes for the three formats SPEC.md 4.2 recommends. +IMAGE_MAGIC = { + b"GIF87a": "GIF", + b"GIF89a": "GIF", + b"\x89PNG\r\n\x1a\n": "PNG", + b"\xff\xd8\xff": "JPEG", +} + + +@dataclass(frozen=True) +class Check: + """One thing the validator looks at, and the spec section it comes from.""" + + code: str + section: str + level: str + what: str + + +CHECKS: tuple[Check, ...] = ( + Check("xml-declaration", "3.1", "must", "UTF-8 XML declaration, no BOM"), + Check("doctype", "3.1", "must", "the XHTML-MP 1.2 doctype, exactly"), + Check("internal-subset", "3.1", "must", "no internal DTD subset"), + Check("well-formed", "3", "must", "the page parses as XML"), + Check("root-element", "3.1", "must", "html root in the XHTML namespace"), + Check("lang", "3.1", "should", "xml:lang and lang agree on html"), + Check("dtd-element", "4.1", "must", "only elements from 4.1"), + Check("dtd-attribute", "4.1", "must", "only attributes from 4.1"), + Check("dtd-nesting", "4.1", "must", "elements nest as the DTD allows"), + Check("dtd-content", "4.1", "must", "element content follows the DTD"), + Check("dtd-value", "4.1", "must", "attribute values from the permitted set"), + Check("dtd-required-attribute", "4.1", "must", "required attributes present"), + Check("dtd-head", "3.3", "must", "the head holds only what 3.3 permits"), + Check("dtd-nested-table", "4.2", "must", "tables are not nested"), + Check("dtd-variant", "5.3", "must", "body class is one of the variants"), + Check("dtd-styling", "5", "must", "no author styling"), + Check("dtd-other", "4.1", "must", "the page matches the Mews DTD"), + Check("marker", "3.2", "must", "the conformance marker is present once"), + Check("marker-version", "3.2", "must", "the marker names a known version"), + Check("meta-name", "3.3", "must", "only the meta names 3.3 permits"), + Check("link-rel", "3.3", "must", "only the link kinds 3.3 permits"), + Check("link-stylesheet-count", "3.3", "must", "at most one stylesheet link"), + Check("link-alternate-type", "3.3", "must", "feed links are Atom"), + Check("icon-offsite", "3.3", "must", "the icon is on the page's own site"), + Check("viewport", "3.3", "should", "the viewport meta is present"), + Check("size-must", "4.3", "must", "markup under 256 KB"), + Check("size-should", "4.3", "should", "markup under 64 KB"), + Check("size-total", "4.3", "should", "page plus images under 320 KB"), + Check("image-data-uri", "4.2", "must", "no data: URI images"), + Check("image-offsite", "4.2", "must", "images are on the page's own site"), + Check("image-format", "4.2", "should", "images are GIF, JPEG or PNG"), + Check("image-size", "4.2", "should", "each image under 50 KB"), + Check("image-unreachable", "4.2", "should", "images load"), + Check("image-metadata", "4.2", "should", "images carry no camera metadata"), + Check("images-not-checked", "4.2", "should", "how many images were checked"), + Check("stylesheet-missing", "5.2", "should", "the default stylesheet is linked"), + Check("stylesheet-canonical", "5.2", "should", "the site hosts its own copy"), + Check("stylesheet-modified", "5.2", "must", "the copy is unmodified"), + Check("stylesheet-import", "5.2", "must", "the copy has no @import"), + Check("stylesheet-unreachable", "5.2", "should", "the stylesheet loads"), + Check("font-offsite", "5.2", "must", "@font-face fonts are on the own site"), + Check("stylesheet-offsite", "7.4", "must", "the stylesheet is on the own site"), + Check("content-type", "7.1", "should", "a page content type"), + Check("https", "7.2", "should", "served over HTTPS"), + Check("validators", "7.3", "should", "Last-Modified or ETag is sent"), + Check("cookie", "7.4", "must", "no cookies are set"), +) + +BY_CODE = {check.code: check for check in CHECKS} + +# Rules in sections 3 to 7 this validator deliberately leaves alone. Together +# with CHECKS this covers every MUST and SHOULD in those sections; tests fail if +# a section appears in neither. +NOT_CHECKED = { + "5.1": "The default stylesheet is compared byte for byte (5.2), which " + "covers whether its base rules are valid WAP CSS.", + "6.1": "Dated links are a client convention; the validator reads no feeds.", + "6.2": "Atom feed contents are a client convention; only the link type is checked.", + "7.2": "Whether a site redirects HTTP to HTTPS in a way old handsets can " + "follow cannot be told from one request.", + "4.2": "Whether text appears only inside an image cannot be told from markup.", +} + + +@dataclass(frozen=True) +class Finding: + """One rule a page broke.""" + + code: str + message: str + location: str = "" + + @property + def section(self) -> str: + """Return the spec section this finding comes from.""" + return BY_CODE[self.code].section + + @property + def level(self) -> str: + """Return either must or should.""" + return BY_CODE[self.code].level + + def __str__(self) -> str: + where = f" ({self.location})" if self.location else "" + return f"Section {self.section} — {self.message}{where}" + + +@dataclass +class Report: + """What the validator found, plus the facts the directory needs.""" + + url: str | None = None + findings: list[Finding] = field(default_factory=list) + notes: list[str] = field(default_factory=list) + title: str = "" + description: str = "" + language: str = "" + size: int = 0 + + def add(self, code: str, message: str, location: str = "") -> None: + """Record one finding, up to the report cap.""" + if len(self.findings) < MAX_FINDINGS: + self.findings.append(Finding(code, message, location)) + + @property + def failures(self) -> list[Finding]: + """Return the findings that make the page non-conforming.""" + return [f for f in self.findings if f.level == "must"] + + @property + def warnings(self) -> list[Finding]: + """Return the findings worth fixing that still leave the page conforming.""" + return [f for f in self.findings if f.level == "should"] + + @property + def conforms(self) -> bool: + """Say whether the page follows every MUST rule the validator checks.""" + return not self.failures + + +def _entity_declarations() -> str: + """Declare the entity set the XHTML-MP doctype would have defined. + + SPEC.md 9.1 says clients must not fetch the doctype, and the local copy of + the driver DTD pulls its modules from w3.org, so the entities are supplied + from the stdlib's HTML 4 table instead — the same 252 names. + """ + return "".join( + f'' + for code, name in sorted(html.entities.codepoint2name.items()) + ) + + +class _Resolver(etree.Resolver): + """Resolves the XHTML-MP doctype to entity declarations and nothing else.""" + + def __init__(self) -> None: + self._entities = _entity_declarations() + + def resolve(self, system_url, public_id, context): + """Answer the parser's request for an external entity.""" + if public_id == "-//WAPFORUM//DTD XHTML Mobile 1.2//EN": + return self.resolve_string(self._entities, context) + # With no_network set, returning None makes any other doctype fail. + return None + + +def _parser(*, expand_entities: bool) -> etree.XMLParser: + """Build an XML parser that never reaches the network.""" + parser = etree.XMLParser( + load_dtd=expand_entities, + resolve_entities=expand_entities, + no_network=True, + dtd_validation=False, + attribute_defaults=False, + huge_tree=False, + ) + if expand_entities: + parser.resolvers.add(_Resolver()) + return parser + + +_DTD: etree.DTD | None = None + + +def mews_dtd() -> etree.DTD: + """Return the Mews DTD, loaded once per process.""" + global _DTD # noqa: PLW0603 + if _DTD is None: + with resources.as_file( + resources.files("mews.data").joinpath("mews-0.1.dtd") + ) as path: + _DTD = etree.DTD(str(path)) + return _DTD + + +def local(element: etree._Element) -> str: + """Return an element's name without its namespace.""" + return etree.QName(element).localname + + +# --- The DTD check ------------------------------------------------------- + +UNDECLARED_ELEMENT = re.compile(r"^No declaration for element (\S+)") +UNDECLARED_ATTRIBUTE = re.compile( + r"^No declaration for attribute (\S+) of element (\S+)" +) +NOT_ALLOWED_IN = re.compile( + r"^Element (\S+) is not declared in (\S+) list of possible children" +) +CONTENT_MODEL = re.compile( + r"^Element (\S+) content does not follow the DTD, expecting (.*?), got \(?(.*?)\)?$" +) +BAD_VALUE = re.compile( + r'^Value "(.*?)" for attribute (\S+) of (\S+) is not among the enumerated set' +) +MISSING_ATTRIBUTE = re.compile(r"^Element (\S+) does not carry attribute (\S+)") + + +def _enumerated_values(element: str, attribute: str) -> list[str]: + """Return the values the DTD permits for an attribute, read from the DTD.""" + for declared in mews_dtd().iterelements(): + if declared.name != element: + continue + for attr in declared.iterattributes(): + if attr.name == attribute: + return list(attr.itervalues() or []) + return [] + + +def _dtd_findings(report: Report, tree: etree._ElementTree) -> None: + """Validate against the Mews DTD and turn libxml2's errors into findings.""" + dtd = mews_dtd() + if dtd.validate(tree): + return + + errors = [(entry.line, entry.message) for entry in dtd.error_log] + undeclared = { + match.group(1) + for _, message in errors + for match in [UNDECLARED_ELEMENT.match(message)] + if match + } + + for line, message in errors: + finding = _map_error(message, undeclared) + if finding is not None: + code, text = finding + report.add(code, text, f"line {line}") + + +def _map_error(message: str, undeclared: set[str]) -> tuple[str, str] | None: + """Map one libxml2 message to a code and a plain sentence, or drop it. + + One mistake makes libxml2 say several things: an unknown element is also an + unknown attribute and a content-model break. Only the clearest is kept. + """ + match = UNDECLARED_ELEMENT.match(message) + if match: + name = match.group(1) + if name == "style": + return ( + "dtd-styling", + "The