feat: page validator for both section 9.1 checks

This commit is contained in:
randogoth 2026-10-11 15:05:38 +03:00
parent 99dc9f7a88
commit b3c436bf1a
13 changed files with 2703 additions and 0 deletions

6
mews/__init__.py Normal file
View file

@ -0,0 +1,6 @@
"""Page validator and site directory for the Mews Profile.
See doc/SPEC.md for the format this package checks.
"""
VERSION = "0.1"

136
mews/css.py Normal file
View file

@ -0,0 +1,136 @@
"""Comparing a site's stylesheet against the default one.
SPEC.md 5.2 lets a site host its own copy of the default stylesheet and change
nothing in it, except to add @font-face rules that load fonts from the site
itself. Checking that means comparing text, since a site may legitimately
differ in line endings and trailing whitespace after a copy-paste.
"""
from dataclasses import dataclass, field
import difflib
from functools import cache
from importlib import resources
import re
FONT_FACE = re.compile(r"@font-face\b", re.IGNORECASE)
URL_IN_SRC = re.compile(r"url\(\s*['\"]?([^'\")]+)", re.IGNORECASE)
IMPORT_RULE = re.compile(r"@import\b", re.IGNORECASE)
@dataclass
class Comparison:
"""What a candidate stylesheet differs from the default by."""
equal: bool
diff_line: int | None = None
diff: str = ""
font_faces: list[str] = field(default_factory=list)
has_import: bool = False
@property
def font_urls(self) -> list[str]:
"""Every url() the added @font-face rules load."""
return [
match.group(1).strip()
for block in self.font_faces
for match in URL_IN_SRC.finditer(block)
]
@cache
def canonical() -> str:
"""Return the normalised default stylesheet shipped with this package."""
data = resources.files("mews.data").joinpath("mews-0.1.css").read_text("utf-8")
return normalise(data)
def normalise(text: str) -> str:
"""Strip the differences a copy may pick up without changing any rule."""
text = text.lstrip("").replace("\r\n", "\n").replace("\r", "\n")
lines = [line.rstrip() for line in text.split("\n")]
out: list[str] = []
for line in lines:
if line or (out and out[-1]):
out.append(line)
return "\n".join(out).strip("\n")
def strip_font_faces(text: str) -> tuple[str, list[str]]:
"""Remove top-level @font-face blocks, returning the rest and the blocks.
Only brace depth zero counts, so an @font-face inside a media query is left
in place and shows up as a difference.
"""
blocks: list[str] = []
out: list[str] = []
index = 0
depth = 0
while index < len(text):
if depth == 0 and FONT_FACE.match(text, index):
end = _block_end(text, index)
if end is None:
break
blocks.append(text[index:end])
index = end
continue
character = text[index]
if character == "{":
depth += 1
elif character == "}":
depth = max(0, depth - 1)
out.append(character)
index += 1
out.append(text[index:])
return "".join(out), blocks
def _block_end(text: str, start: int) -> int | None:
"""Index just past the brace-balanced block beginning at start."""
opened = text.find("{", start)
if opened == -1:
return None
depth = 0
for index in range(opened, len(text)):
if text[index] == "{":
depth += 1
elif text[index] == "}":
depth -= 1
if depth == 0:
return index + 1
return None
def compare(candidate: str) -> Comparison:
"""Compare a stylesheet against the default, allowing added @font-face rules."""
expected = canonical()
if normalise(candidate) == expected:
return Comparison(equal=True)
remainder, blocks = strip_font_faces(candidate)
found = normalise(remainder)
has_import = bool(IMPORT_RULE.search(candidate))
if found == expected:
return Comparison(equal=True, font_faces=blocks, has_import=has_import)
diff = list(
difflib.unified_diff(
expected.split("\n"), found.split("\n"), "mews-0.1.css", "your copy", n=1
)
)
line = next(
(
index + 1
for index, (left, right) in enumerate(
zip(expected.split("\n"), found.split("\n"), strict=False)
)
if left != right
),
min(len(expected.split("\n")), len(found.split("\n"))) + 1,
)
return Comparison(
equal=False,
diff_line=line,
diff="\n".join(diff[:8]),
font_faces=blocks,
has_import=has_import,
)

7
mews/data/README.md Normal file
View file

@ -0,0 +1,7 @@
# Packaged copies
The canonical files live at the repository root (`mews-0.1.css`) and in `dtd/`
(`mews-0.1.dtd`), where SPEC.md and `dtd/check_dtd.py` reference them. These
copies exist so the validator still finds them when it is installed as a wheel,
with nothing but the package on disk. `tests/test_data.py` fails if a copy and
its original drift apart.

106
mews/data/mews-0.1.css Normal file
View file

@ -0,0 +1,106 @@
/* Mews Profile default stylesheet 0.1 */
/* Base: light theme */
body {
background-color: #faf8f3;
color: #1f1f1f;
font-family: "Atkinson Hyperlegible", Verdana, Tahoma, system-ui, sans-serif;
font-size: 106%;
line-height: 1.5;
margin: 0 auto;
padding: 1em;
max-width: 40em;
}
h1, h2, h3, h4, h5, h6 {
font-weight: bold;
line-height: 1.25;
margin: 1.6em 0 0.5em 0;
}
h1 { font-size: 1.6em; margin-top: 0.5em; }
h2 { font-size: 1.3em; }
h3 { font-size: 1.1em; }
h4, h5, h6 { font-size: 1em; }
p, ul, ol, dl, blockquote, pre, table, address {
margin: 0 0 1em 0;
}
ul, ol { padding-left: 1.5em; }
li { margin-bottom: 0.25em; }
dt { font-weight: bold; }
dd { margin: 0 0 0.5em 1.5em; }
a { color: #1a4f9c; text-decoration: underline; }
a:visited { color: #5b2a86; }
a:focus, a:active { outline: 2px solid #1a4f9c; }
blockquote {
margin-left: 0;
padding-left: 1em;
border-left: 3px solid #c9c4b8;
}
pre, code, kbd, samp {
font-family: "Source Code Pro", Consolas, Menlo, "DejaVu Sans Mono", monospace;
font-size: 0.9em;
}
pre {
background-color: #efece4;
padding: 0.75em;
overflow: auto;
}
hr {
border: 0;
border-top: 1px solid #c9c4b8;
margin: 2em 0;
}
table { border-collapse: collapse; }
th, td {
border: 1px solid #c9c4b8;
padding: 0.3em 0.6em;
text-align: left;
vertical-align: top;
}
caption { font-weight: bold; text-align: left; }
img {
display: block;
max-width: 100%;
height: auto;
border: 0;
}
input, select, textarea { font-size: 1em; font-family: inherit; }
/* Body variants: light */
body.mews-warm { background-color: #fbf4e8; color: #2a2420; }
body.mews-warm a { color: #8a3d14; }
body.mews-cool { background-color: #f3f6fa; color: #1c2430; }
body.mews-cool a { color: #1a4f9c; }
body.mews-green { background-color: #f2f6ef; color: #1f2a1f; }
body.mews-green a { color: #24502a; }
body.mews-mono { background-color: #e8ebe1; color: #1e2219; }
body.mews-mono a { color: #1e2219; }
/* Dark theme: follows the system setting */
@media (prefers-color-scheme: dark) {
body { background-color: #18191b; color: #e3e1dc; }
a { color: #8fb8f5; }
a:visited { color: #c9a3f2; }
a:focus, a:active { outline-color: #8fb8f5; }
blockquote, hr, th, td { border-color: #45464a; }
pre { background-color: #222326; }
body.mews-warm { background-color: #1d1916; color: #e8dfd3; }
body.mews-warm a { color: #f0b48a; }
body.mews-cool { background-color: #161a20; color: #dde4ee; }
body.mews-cool a { color: #8fb8f5; }
body.mews-green { background-color: #161b16; color: #dbe6d8; }
body.mews-green a { color: #9fd39a; }
body.mews-mono { background-color: #1b1d18; color: #c8d0b8; }
body.mews-mono a { color: #c8d0b8; }
}
html { color-scheme: light dark; }

396
mews/data/mews-0.1.dtd Normal file
View file

@ -0,0 +1,396 @@
<!-- Mews Profile 0.1 DTD
Published at mews.page/dtd/mews-0.1.dtd. Browsers and clients must not
fetch this file.
XHTML Mobile Profile 1.2 restricted as in the Mews Profile
Specification 0.1 section 4, extended with lang and dir (section 3).
Maintained by hand and checked against the XHTML Mobile 1.2 driver,
xhtml-mobile12.dtd (OMA-SUP-DTD_xhtml_mobile12-V1_2-20080331-A), by
check_dtd.py. It cannot be generated by stripping the driver: the
driver declares no elements itself, attribute lists only ever add,
and the rules below do not exist in XHTML-MP to be kept.
This file is standalone: it declares every element and attribute
itself, so validation never depends on the Open Mobile Alliance's or
W3C's servers. Pages keep the XHTML-MP doctype (section 3.1); only
validators use this file (section 9.1).
Encoded here: the permitted elements and attributes (4.1), the head
rules (3.3), the body variant classes (5.3) and the ban on nested
tables (4.2). 51 elements are kept from XHTML Mobile 1.2. Dropped:
acronym, base, big, button, div, legend, noscript, object, param,
script, small, span, style, sub, sup, tt. Event attributes and the
style attribute are omitted entirely.
-->
<!-- Datatypes (as in XHTML Modularization) -->
<!ENTITY % Text.datatype "CDATA" >
<!ENTITY % URI.datatype "CDATA" >
<!ENTITY % Number.datatype "CDATA" >
<!ENTITY % LanguageCode.datatype "NMTOKEN" >
<!ENTITY % Character.datatype "CDATA" >
<!-- Common attributes. id, title, xml:lang, lang and dir are permitted
on every element in the body (SPEC.md 4.1); lang and dir are the Mews
extensions over XHTML-MP (SPEC.md 3). -->
<!ENTITY % id.attrib "id ID #IMPLIED" >
<!ENTITY % title.attrib "title CDATA #IMPLIED" >
<!ENTITY % i18n.attrib
"xml:lang %LanguageCode.datatype; #IMPLIED
lang %LanguageCode.datatype; #IMPLIED
dir (ltr | rtl) #IMPLIED" >
<!ENTITY % Common.attrib "%id.attrib; %title.attrib; %i18n.attrib;" >
<!-- Inline classes -->
<!ENTITY % InlStruct.class "br" >
<!ENTITY % InlPhras.class "| em | strong | code | kbd | samp | var | q | cite | abbr | dfn" >
<!ENTITY % InlPres.class "| b | i" >
<!ENTITY % Anchor.class "| a" >
<!ENTITY % InlSpecial.class "| img" >
<!ENTITY % InlFormNoLabel.class "| input | select | textarea" >
<!ENTITY % InlForm.class "%InlFormNoLabel.class; | label" >
<!ENTITY % Inline.extra "" >
<!ENTITY % Inline.class "%InlStruct.class;
%InlPhras.class;
%InlPres.class;
%Anchor.class;
%InlSpecial.class;
%InlForm.class;
%Inline.extra;" >
<!ENTITY % InlNoAnchor.class "%InlStruct.class;
%InlPhras.class;
%InlPres.class;
%InlSpecial.class;
%InlForm.class;
%Inline.extra;" >
<!ENTITY % InlNoLabel.class "%InlStruct.class;
%InlPhras.class;
%InlPres.class;
%Anchor.class;
%InlSpecial.class;
%InlFormNoLabel.class;
%Inline.extra;" >
<!-- Block classes -->
<!ENTITY % Heading.class "h1 | h2 | h3 | h4 | h5 | h6" >
<!ENTITY % List.class "ul | ol | dl" >
<!ENTITY % BlkStruct.class "p" >
<!ENTITY % BlkPhras.class "| pre | blockquote | address" >
<!ENTITY % BlkPres.class "| hr" >
<!ENTITY % Table.class "| table" >
<!ENTITY % Form.class "| form" >
<!ENTITY % Fieldset.class "| fieldset" >
<!ENTITY % Block.extra "" >
<!ENTITY % BlkNoForm.class "%Heading.class;
| %List.class;
| %BlkStruct.class;
%BlkPhras.class;
%BlkPres.class;
%Fieldset.class;
%Block.extra;" >
<!ENTITY % BlkNoTable.class "%BlkNoForm.class;
%Form.class;
%Block.extra;" >
<!ENTITY % Blk.class "%BlkNoTable.class;
%Table.class;
%Block.extra;" >
<!ENTITY % FlowNoTable.class "%Inline.class;
| %BlkNoTable.class;" >
<!-- pre content model, taken verbatim from the driver. It excludes
%InlPres.class; and %InlSpecial.class;, so b, i and img are not
permitted in pre, as in XHTML Mobile 1.2. -->
<!ENTITY % pre.content
"( #PCDATA
| %InlStruct.class;
%InlPhras.class;
%Anchor.class;
%Inline.extra; )*"
>
<!-- Document structure (SPEC.md 4.1) -->
<!--
Only body uses %Blk.class;, so a table can appear directly in the
body and nowhere else, and tables cannot nest (SPEC.md 4.2). body may
carry one class from the variant list (SPEC.md 5.3).
-->
<!ELEMENT html ( head, body ) >
<!ATTLIST html
xmlns CDATA #FIXED "http://www.w3.org/1999/xhtml"
xml:lang %LanguageCode.datatype; #IMPLIED
lang %LanguageCode.datatype; #IMPLIED
dir (ltr | rtl) #IMPLIED
>
<!ELEMENT head ( title, meta, ( meta | link )* ) >
<!ELEMENT title ( #PCDATA )* >
<!ELEMENT body ( %Blk.class; )+ >
<!ATTLIST body
%Common.attrib;
class (mews-warm | mews-cool | mews-green | mews-mono) #IMPLIED
>
<!-- Head elements (SPEC.md 3.3) -->
<!--
The head may contain only title, meta and link. The title comes
first and at least one meta is required; the conformance marker is a
meta (SPEC.md 3.2), but a DTD cannot check its name and content
together, so validators confirm the marker as a rule check (SPEC.md
9.1). link rel is limited to the uses SPEC.md 3.3 names: the
stylesheet, the icon and feed links.
-->
<!ELEMENT meta EMPTY >
<!ATTLIST meta
name NMTOKEN #IMPLIED
content CDATA #REQUIRED
>
<!ELEMENT link EMPTY >
<!ATTLIST link
rel (alternate | icon | stylesheet) #IMPLIED
href %URI.datatype; #IMPLIED
type CDATA #IMPLIED
title CDATA #IMPLIED
>
<!-- Headings (SPEC.md 4.1) -->
<!ELEMENT h1 ( #PCDATA | %Inline.class; )* >
<!ATTLIST h1
%Common.attrib;
>
<!ELEMENT h2 ( #PCDATA | %Inline.class; )* >
<!ATTLIST h2
%Common.attrib;
>
<!ELEMENT h3 ( #PCDATA | %Inline.class; )* >
<!ATTLIST h3
%Common.attrib;
>
<!ELEMENT h4 ( #PCDATA | %Inline.class; )* >
<!ATTLIST h4
%Common.attrib;
>
<!ELEMENT h5 ( #PCDATA | %Inline.class; )* >
<!ATTLIST h5
%Common.attrib;
>
<!ELEMENT h6 ( #PCDATA | %Inline.class; )* >
<!ATTLIST h6
%Common.attrib;
>
<!-- Text blocks (SPEC.md 4.1) -->
<!ELEMENT p ( #PCDATA | %Inline.class; )* >
<!ATTLIST p
%Common.attrib;
>
<!ELEMENT blockquote ( %BlkNoTable.class; )* >
<!ATTLIST blockquote
%Common.attrib;
cite %URI.datatype; #IMPLIED
>
<!ELEMENT pre %pre.content; >
<!ATTLIST pre
%Common.attrib;
>
<!ELEMENT address ( #PCDATA | %Inline.class; )* >
<!ATTLIST address
%Common.attrib;
>
<!ELEMENT hr EMPTY >
<!ATTLIST hr
%Common.attrib;
>
<!-- Inline text (SPEC.md 4.1) -->
<!ELEMENT em ( #PCDATA | %Inline.class; )* >
<!ATTLIST em
%Common.attrib;
>
<!ELEMENT strong ( #PCDATA | %Inline.class; )* >
<!ATTLIST strong
%Common.attrib;
>
<!ELEMENT code ( #PCDATA | %Inline.class; )* >
<!ATTLIST code
%Common.attrib;
>
<!ELEMENT kbd ( #PCDATA | %Inline.class; )* >
<!ATTLIST kbd
%Common.attrib;
>
<!ELEMENT samp ( #PCDATA | %Inline.class; )* >
<!ATTLIST samp
%Common.attrib;
>
<!ELEMENT var ( #PCDATA | %Inline.class; )* >
<!ATTLIST var
%Common.attrib;
>
<!ELEMENT cite ( #PCDATA | %Inline.class; )* >
<!ATTLIST cite
%Common.attrib;
>
<!ELEMENT abbr ( #PCDATA | %Inline.class; )* >
<!ATTLIST abbr
%Common.attrib;
>
<!ELEMENT dfn ( #PCDATA | %Inline.class; )* >
<!ATTLIST dfn
%Common.attrib;
>
<!ELEMENT b ( #PCDATA | %Inline.class; )* >
<!ATTLIST b
%Common.attrib;
>
<!ELEMENT i ( #PCDATA | %Inline.class; )* >
<!ATTLIST i
%Common.attrib;
>
<!ELEMENT q ( #PCDATA | %Inline.class; )* >
<!ATTLIST q
%Common.attrib;
cite %URI.datatype; #IMPLIED
>
<!ELEMENT br EMPTY >
<!ATTLIST br
%Common.attrib;
>
<!ELEMENT a ( #PCDATA | %InlNoAnchor.class; )* >
<!ATTLIST a
%Common.attrib;
href %URI.datatype; #IMPLIED
hreflang %LanguageCode.datatype; #IMPLIED
rel CDATA #IMPLIED
type CDATA #IMPLIED
accesskey %Character.datatype; #IMPLIED
>
<!ELEMENT img EMPTY >
<!ATTLIST img
%Common.attrib;
src %URI.datatype; #REQUIRED
alt %Text.datatype; #REQUIRED
>
<!-- Lists (SPEC.md 4.1) -->
<!ELEMENT ul ( li )+ >
<!ATTLIST ul
%Common.attrib;
>
<!ELEMENT ol ( li )+ >
<!ATTLIST ol
%Common.attrib;
start %Number.datatype; #IMPLIED
>
<!ELEMENT dl ( dt | dd )+ >
<!ATTLIST dl
%Common.attrib;
>
<!ELEMENT li ( #PCDATA | %FlowNoTable.class; )* >
<!ATTLIST li
%Common.attrib;
value %Number.datatype; #IMPLIED
>
<!ELEMENT dt ( #PCDATA | %Inline.class; )* >
<!ATTLIST dt
%Common.attrib;
>
<!ELEMENT dd ( #PCDATA | %FlowNoTable.class; )* >
<!ATTLIST dd
%Common.attrib;
>
<!-- Tables (SPEC.md 4.2) -->
<!--
As in the XHTML Mobile basic tables module, cells take
%FlowNoTable.class; and never a table, so tables cannot nest
(SPEC.md 4.2).
-->
<!ELEMENT table ( caption?, tr+ ) >
<!ATTLIST table
%Common.attrib;
summary %Text.datatype; #IMPLIED
>
<!ELEMENT caption ( #PCDATA | %Inline.class; )* >
<!ATTLIST caption
%Common.attrib;
>
<!ELEMENT tr ( th | td )+ >
<!ATTLIST tr
%Common.attrib;
>
<!ELEMENT th ( #PCDATA | %FlowNoTable.class; )* >
<!ATTLIST th
%Common.attrib;
rowspan %Number.datatype; "1"
colspan %Number.datatype; "1"
scope (row | col) #IMPLIED
>
<!ELEMENT td ( #PCDATA | %FlowNoTable.class; )* >
<!ATTLIST td
%Common.attrib;
rowspan %Number.datatype; "1"
colspan %Number.datatype; "1"
scope (row | col) #IMPLIED
>
<!-- Forms (SPEC.md 4.1, 4.2) -->
<!--
As in the XHTML MP forms module, form takes %BlkNoForm.class;, so
forms cannot nest. fieldset also takes %BlkNoForm.class;, so a
fieldset cannot reintroduce a form.
-->
<!ELEMENT form ( %BlkNoForm.class; )+ >
<!ATTLIST form
%Common.attrib;
action %URI.datatype; #REQUIRED
method (get | post) "get"
enctype CDATA "application/x-www-form-urlencoded"
>
<!ELEMENT fieldset ( #PCDATA | %Inline.class; | %BlkNoForm.class; )* >
<!ATTLIST fieldset
%Common.attrib;
>
<!ELEMENT label ( #PCDATA | %InlNoLabel.class; )* >
<!ATTLIST label
%Common.attrib;
for IDREF #IMPLIED
accesskey %Character.datatype; #IMPLIED
>
<!ELEMENT input EMPTY >
<!ATTLIST input
%Common.attrib;
type (text | password | checkbox | radio | submit | reset | hidden) "text"
name CDATA #IMPLIED
value CDATA #IMPLIED
checked (checked) #IMPLIED
size %Number.datatype; #IMPLIED
maxlength %Number.datatype; #IMPLIED
accesskey %Character.datatype; #IMPLIED
inputmode CDATA #IMPLIED
>
<!ELEMENT select ( optgroup | option )+ >
<!ATTLIST select
%Common.attrib;
name CDATA #IMPLIED
multiple (multiple) #IMPLIED
size %Number.datatype; #IMPLIED
>
<!ELEMENT optgroup ( option )+ >
<!ATTLIST optgroup
%Common.attrib;
label %Text.datatype; #REQUIRED
>
<!ELEMENT option ( #PCDATA )* >
<!ATTLIST option
%Common.attrib;
value CDATA #IMPLIED
selected (selected) #IMPLIED
>
<!ELEMENT textarea ( #PCDATA )* >
<!ATTLIST textarea
%Common.attrib;
name CDATA #IMPLIED
rows %Number.datatype; #REQUIRED
cols %Number.datatype; #REQUIRED
accesskey %Character.datatype; #IMPLIED
inputmode CDATA #IMPLIED
>

382
mews/fetch.py Normal file
View file

@ -0,0 +1,382 @@
"""Fetching pages that strangers ask the checker to look at.
Every URL reaching this module is author-supplied, so the service must not be
usable as a probe against the machine it runs on or its neighbours. Addresses
are vetted before the connection and the connection is pinned to the vetted
address, so a second DNS answer cannot redirect it (DNS rebinding).
"""
from dataclasses import dataclass, replace
import ipaddress
import os
import re
import socket
import time
from urllib.parse import urlsplit, urlunsplit
import httpx
from publicsuffixlist import PublicSuffixList
# Read caps, in bytes. The page cap is above the 256 KB that SPEC.md 4.3 makes a
# MUST so an oversized page is reported as too large rather than unreachable.
PAGE_CAP = 288 * 1024
IMAGE_CAP = 64 * 1024
CSS_CAP = 256 * 1024
MAX_REDIRECTS = 3
MAX_URL_LENGTH = 2048
USER_AGENT = "mews.page checker (+https://mews.page/about)"
_PSL = PublicSuffixList()
# Ranges ipaddress does not classify but which must never be reached: carrier
# NAT (which is also Tailscale's range), IETF protocol assignment, benchmarking,
# the documentation ranges, and the IPv6 transition mechanisms that tunnel an
# IPv4 address inside an address that looks global.
_BLOCKED_NETS = [
ipaddress.ip_network(net)
for net in (
"100.64.0.0/10",
"192.0.0.0/24",
"198.18.0.0/15",
"192.0.2.0/24",
"198.51.100.0/24",
"203.0.113.0/24",
"240.0.0.0/4",
"255.255.255.255/32",
"2001:db8::/32",
"2002::/16",
"2001::/32",
"64:ff9b::/96",
)
]
# Hosts whose own addresses must be unreachable, as CIDRs. The deployment sets
# this to the machine's public address so a submission cannot be aimed at a
# service sharing the box.
_EXTRA_NETS = [
ipaddress.ip_network(net.strip())
for net in os.environ.get("MEWS_BLOCK_NETS", "").split(",")
if net.strip()
]
_LOCAL_SUFFIXES = (".local", ".internal", ".home.arpa", ".localhost")
# What a host name may be made of. Checking this first means a typo gets a
# clearer answer than a complaint about public suffixes.
_HOSTNAME = re.compile(
r"^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)*$"
)
class UrlError(Exception):
"""A URL the checker will not fetch. The message is shown to the author."""
class FetchError(Exception):
"""A fetch that produced no page. The message is shown to the author."""
@dataclass(frozen=True)
class Fetched:
"""One response the checker read, with the body it kept."""
url: str
status: int
headers: httpx.Headers
body: bytes
truncated: bool
scheme_downgraded: bool = False
def registered_domain(host: str) -> str | None:
"""Return the registrable domain of host, or None when it has no public suffix.
Comparing registrable domains rather than hostnames is what makes the
same-site rules in SPEC.md 4.2 and 7.4 mean anything: a free subdomain host
would otherwise let any two unrelated sites count as one.
"""
if not host:
return None
try:
name = host.strip().rstrip(".").lower().encode("idna").decode("ascii")
except (UnicodeError, UnicodeDecodeError):
return None
return _PSL.privatesuffix(name)
def same_site(a: str, b: str) -> bool:
"""Whether two hosts share a registrable domain."""
first = registered_domain(a)
return first is not None and first == registered_domain(b)
def normalise_url(raw: str, *, allow_loopback: bool = False) -> str:
"""Return a URL the checker is willing to fetch, or raise UrlError.
This runs before any name lookup, so a rejected URL costs nothing. With
allow_loopback the rules relax enough to reach a test server on this
machine: an address literal, localhost, and any port.
"""
text = (raw or "").strip()
if not text:
raise UrlError("Enter the address of a page to check.")
if len(text) > MAX_URL_LENGTH:
raise UrlError("That address is too long.")
if "://" not in text:
text = "https://" + text
parts = urlsplit(text)
if parts.scheme not in ("http", "https"):
raise UrlError("Enter an address that starts with http:// or https://.")
if "@" in parts.netloc:
raise UrlError("Enter an address without a user name in it.")
try:
host, port = parts.hostname, parts.port
except ValueError as error:
raise UrlError("That address has a port the checker can't read.") from error
if not host:
raise UrlError(
"That doesn't look like a web address. Enter the full address "
"of a page, like https://example.com/"
)
if port is not None and port not in (80, 443) and not allow_loopback:
raise UrlError("The checker reads pages on the usual web ports only.")
host = host.lower().rstrip(".")
if _is_ip_literal(host):
if not (allow_loopback and _is_loopback_literal(host)):
raise UrlError("Enter a domain name rather than an IP address.")
elif host == "localhost" or host.endswith(_LOCAL_SUFFIXES):
if not allow_loopback:
raise UrlError("That address is only reachable on a local network.")
elif not _HOSTNAME.match(_ascii(host)):
raise UrlError(
"That doesn't look like a web address. Enter the full address "
"of a page, like https://example.com/"
)
elif registered_domain(host) is None:
raise UrlError("That domain name isn't one the checker can reach.")
netloc = host if port is None else f"{host}:{port}"
return urlunsplit((parts.scheme, netloc, parts.path or "/", parts.query, ""))
def _ascii(host: str) -> str:
"""Return the punycode form of a host name, or the name unchanged."""
try:
return host.encode("idna").decode("ascii")
except (UnicodeError, UnicodeDecodeError):
return host
def _is_ip_literal(host: str) -> bool:
"""Whether host is an address literal in any of the forms a parser accepts."""
candidate = host.strip("[]")
try:
ipaddress.ip_address(candidate)
except ValueError:
pass
else:
return True
# Decimal, octal and hex integer forms of an IPv4 address, which urlsplit
# leaves alone but a resolver would accept.
if host.isdigit():
return True
bare = host.replace(".", "")
return bare.startswith(("0x", "0X")) or (
host.startswith("0") and len(host) > 1 and bare.isdigit()
)
def _is_loopback_literal(host: str) -> bool:
"""Whether host is a literal address on this machine."""
try:
return ipaddress.ip_address(host.strip("[]")).is_loopback
except ValueError:
return False
def _embedded(address: ipaddress.IPv6Address) -> ipaddress.IPv4Address | None:
"""Return the IPv4 address a transition mechanism hides inside an IPv6 one."""
for attribute in ("ipv4_mapped", "sixtofour"):
value = getattr(address, attribute, None)
if value is not None:
return value
teredo = getattr(address, "teredo", None)
if teredo:
return teredo[1]
if address in ipaddress.ip_network("64:ff9b::/96"):
return ipaddress.IPv4Address(int(address) & 0xFFFFFFFF)
return None
def vet_address(address: str, *, allow_loopback: bool = False) -> None:
"""Raise UrlError unless address is a public one the checker may connect to."""
ip = ipaddress.ip_address(address)
if allow_loopback and ip.is_loopback:
return
if (
ip.is_private
or ip.is_loopback
or ip.is_link_local
or ip.is_reserved
or ip.is_multicast
or ip.is_unspecified
or any(ip in net for net in _BLOCKED_NETS)
or any(ip in net for net in _EXTRA_NETS)
):
raise UrlError("That address isn't on the public internet.")
if isinstance(ip, ipaddress.IPv6Address):
inner = _embedded(ip)
if inner is not None:
vet_address(str(inner), allow_loopback=allow_loopback)
def resolve(url: str, *, allow_loopback: bool = False) -> str:
"""Return one vetted address for the URL's host, rejecting the host if any fails.
Every answer has to pass: a host that resolves to one public and one private
address would otherwise be a coin toss.
"""
parts = urlsplit(url)
host = parts.hostname or ""
port = parts.port or (443 if parts.scheme == "https" else 80)
try:
infos = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM)
except socket.gaierror as error:
raise FetchError(
"We couldn't find that domain name. Check the address and try again."
) from error
addresses = [info[4][0] for info in infos]
if not addresses:
raise FetchError("We couldn't find that domain name.")
for address in addresses:
vet_address(address, allow_loopback=allow_loopback)
return addresses[0]
class Fetcher:
"""Reads pages and their sub-resources, under a time and byte budget.
One Fetcher serves one check, so the whole-check budget is shared across the
page, its stylesheet and its images.
"""
def __init__(self, *, allow_loopback: bool = False, budget: float = 45.0):
self.allow_loopback = allow_loopback
self._deadline = time.monotonic() + budget
self._client = httpx.Client(
follow_redirects=False,
trust_env=False,
http2=False,
verify=True,
timeout=httpx.Timeout(connect=5.0, read=10.0, write=5.0, pool=5.0),
limits=httpx.Limits(max_connections=4, max_keepalive_connections=2),
headers={
"User-Agent": USER_AGENT,
"Accept": "text/html,application/xhtml+xml",
"Accept-Encoding": "gzip",
},
)
def __enter__(self) -> "Fetcher":
return self
def __exit__(self, *exc: object) -> None:
self.close()
def close(self) -> None:
"""Close the underlying connections."""
self._client.close()
def get(
self,
url: str,
*,
cap: int = PAGE_CAP,
headers: dict[str, str] | None = None,
) -> Fetched:
"""Fetch one URL, following redirects by hand and vetting each hop."""
current = normalise_url(url, allow_loopback=self.allow_loopback)
downgraded = False
for _ in range(MAX_REDIRECTS + 1):
response = self._request(current, cap=cap, headers=headers)
if response.status not in (301, 302, 303, 307, 308):
if downgraded:
return replace(response, scheme_downgraded=True)
return response
location = response.headers.get("location", "")
if not location:
raise FetchError("That page redirects without saying where to.")
target = normalise_url(
str(httpx.URL(current).join(location)),
allow_loopback=self.allow_loopback,
)
was_secure = urlsplit(current).scheme == "https"
if was_secure and urlsplit(target).scheme == "http":
downgraded = True
current = target
raise FetchError("That page redirects too many times.")
def _request(
self,
url: str,
*,
cap: int,
headers: dict[str, str] | None,
) -> Fetched:
"""Make one pinned request, streaming the body up to cap bytes."""
self._check_deadline()
parts = urlsplit(url)
host = parts.hostname or ""
port = parts.port
address = resolve(url, allow_loopback=self.allow_loopback)
literal = f"[{address}]" if ":" in address else address
pinned = urlunsplit(
(
parts.scheme,
literal if port is None else f"{literal}:{port}",
parts.path,
parts.query,
"",
)
)
request_headers = {"Host": parts.netloc, **(headers or {})}
extensions = {"sni_hostname": host} if parts.scheme == "https" else {}
try:
with self._client.stream(
"GET", pinned, headers=request_headers, extensions=extensions
) as response:
declared = response.headers.get("content-length")
if declared and declared.isdigit() and int(declared) > cap:
raise FetchError("That page is too large for the checker to read.")
body, truncated = self._read(response, cap)
except httpx.TimeoutException as error:
raise FetchError("That page took too long to answer.") from error
except httpx.HTTPError as error:
raise FetchError("We couldn't connect to that site.") from error
return Fetched(url, response.status_code, response.headers, body, truncated)
def _read(self, response: httpx.Response, cap: int) -> tuple[bytes, bool]:
"""Read a streamed body, stopping at cap decoded bytes."""
chunks: list[bytes] = []
total = 0
for chunk in response.iter_bytes():
self._check_deadline()
chunks.append(chunk)
total += len(chunk)
if total > cap:
return b"".join(chunks)[:cap], True
body = b"".join(chunks)
encoded = response.num_bytes_downloaded or len(body)
# A body that expanded enormously from a small download is a compression
# bomb, not a page.
if encoded and len(body) > 100 * encoded and len(body) > 64 * 1024:
raise FetchError("That page is compressed in a way the checker won't read.")
return body, False
def _check_deadline(self) -> None:
if time.monotonic() > self._deadline:
raise FetchError("Checking that page took too long.")

976
mews/lint.py Normal file
View file

@ -0,0 +1,976 @@
"""The Mews page validator.
SPEC.md 9.1 defines conformance as two checks: the page validates against
dtd/mews-0.1.dtd, and it follows the rules a DTD cannot express. This module
runs both and reports every failure with its spec section.
Failures at MUST level mean the page does not conform. Failures at SHOULD level
are warnings: they are worth fixing but do not make a page non-conforming.
"""
import argparse
from dataclasses import dataclass, field
import html.entities
from importlib import resources
from pathlib import Path
import re
import sys
from urllib.parse import urljoin, urlsplit
from lxml import etree
from mews import VERSION, css as mews_css
from mews.fetch import (
CSS_CAP,
IMAGE_CAP,
Fetched,
Fetcher,
FetchError,
UrlError,
normalise_url,
same_site,
)
XHTML = "http://www.w3.org/1999/xhtml"
XML_LANG = "{http://www.w3.org/XML/1998/namespace}lang"
MARKER = "mews-profile"
CANONICAL_STYLESHEET = "https://mews.page/mews-0.1.css"
OK = 200
MAX_FINDINGS = 50
MAX_IMAGES = 20
SIZE_MUST = 256 * 1024
SIZE_SHOULD = 64 * 1024
SIZE_TOTAL = 320 * 1024
IMAGE_SHOULD = 50 * 1024
XML_DECLARATION = re.compile(rb'^<\?xml version="1\.0" encoding="(?i:UTF-8)"\?>')
DOCTYPE = re.compile(
rb'<!DOCTYPE\s+html\s+PUBLIC\s+"-//WAPFORUM//DTD XHTML Mobile 1\.2//EN"\s+'
rb'"http://www\.openmobilealliance\.org/tech/DTD/xhtml-mobile12\.dtd"\s*>'
)
ALLOWED_META = ("mews-profile", "description", "author", "viewport")
ALLOWED_RELS = ("stylesheet", "icon", "alternate")
# Magic bytes for the three formats SPEC.md 4.2 recommends.
IMAGE_MAGIC = {
b"GIF87a": "GIF",
b"GIF89a": "GIF",
b"\x89PNG\r\n\x1a\n": "PNG",
b"\xff\xd8\xff": "JPEG",
}
@dataclass(frozen=True)
class Check:
"""One thing the validator looks at, and the spec section it comes from."""
code: str
section: str
level: str
what: str
CHECKS: tuple[Check, ...] = (
Check("xml-declaration", "3.1", "must", "UTF-8 XML declaration, no BOM"),
Check("doctype", "3.1", "must", "the XHTML-MP 1.2 doctype, exactly"),
Check("internal-subset", "3.1", "must", "no internal DTD subset"),
Check("well-formed", "3", "must", "the page parses as XML"),
Check("root-element", "3.1", "must", "html root in the XHTML namespace"),
Check("lang", "3.1", "should", "xml:lang and lang agree on html"),
Check("dtd-element", "4.1", "must", "only elements from 4.1"),
Check("dtd-attribute", "4.1", "must", "only attributes from 4.1"),
Check("dtd-nesting", "4.1", "must", "elements nest as the DTD allows"),
Check("dtd-content", "4.1", "must", "element content follows the DTD"),
Check("dtd-value", "4.1", "must", "attribute values from the permitted set"),
Check("dtd-required-attribute", "4.1", "must", "required attributes present"),
Check("dtd-head", "3.3", "must", "the head holds only what 3.3 permits"),
Check("dtd-nested-table", "4.2", "must", "tables are not nested"),
Check("dtd-variant", "5.3", "must", "body class is one of the variants"),
Check("dtd-styling", "5", "must", "no author styling"),
Check("dtd-other", "4.1", "must", "the page matches the Mews DTD"),
Check("marker", "3.2", "must", "the conformance marker is present once"),
Check("marker-version", "3.2", "must", "the marker names a known version"),
Check("meta-name", "3.3", "must", "only the meta names 3.3 permits"),
Check("link-rel", "3.3", "must", "only the link kinds 3.3 permits"),
Check("link-stylesheet-count", "3.3", "must", "at most one stylesheet link"),
Check("link-alternate-type", "3.3", "must", "feed links are Atom"),
Check("icon-offsite", "3.3", "must", "the icon is on the page's own site"),
Check("viewport", "3.3", "should", "the viewport meta is present"),
Check("size-must", "4.3", "must", "markup under 256 KB"),
Check("size-should", "4.3", "should", "markup under 64 KB"),
Check("size-total", "4.3", "should", "page plus images under 320 KB"),
Check("image-data-uri", "4.2", "must", "no data: URI images"),
Check("image-offsite", "4.2", "must", "images are on the page's own site"),
Check("image-format", "4.2", "should", "images are GIF, JPEG or PNG"),
Check("image-size", "4.2", "should", "each image under 50 KB"),
Check("image-unreachable", "4.2", "should", "images load"),
Check("image-metadata", "4.2", "should", "images carry no camera metadata"),
Check("images-not-checked", "4.2", "should", "how many images were checked"),
Check("stylesheet-missing", "5.2", "should", "the default stylesheet is linked"),
Check("stylesheet-canonical", "5.2", "should", "the site hosts its own copy"),
Check("stylesheet-modified", "5.2", "must", "the copy is unmodified"),
Check("stylesheet-import", "5.2", "must", "the copy has no @import"),
Check("stylesheet-unreachable", "5.2", "should", "the stylesheet loads"),
Check("font-offsite", "5.2", "must", "@font-face fonts are on the own site"),
Check("stylesheet-offsite", "7.4", "must", "the stylesheet is on the own site"),
Check("content-type", "7.1", "should", "a page content type"),
Check("https", "7.2", "should", "served over HTTPS"),
Check("validators", "7.3", "should", "Last-Modified or ETag is sent"),
Check("cookie", "7.4", "must", "no cookies are set"),
)
BY_CODE = {check.code: check for check in CHECKS}
# Rules in sections 3 to 7 this validator deliberately leaves alone. Together
# with CHECKS this covers every MUST and SHOULD in those sections; tests fail if
# a section appears in neither.
NOT_CHECKED = {
"5.1": "The default stylesheet is compared byte for byte (5.2), which "
"covers whether its base rules are valid WAP CSS.",
"6.1": "Dated links are a client convention; the validator reads no feeds.",
"6.2": "Atom feed contents are a client convention; only the link type is checked.",
"7.2": "Whether a site redirects HTTP to HTTPS in a way old handsets can "
"follow cannot be told from one request.",
"4.2": "Whether text appears only inside an image cannot be told from markup.",
}
@dataclass(frozen=True)
class Finding:
"""One rule a page broke."""
code: str
message: str
location: str = ""
@property
def section(self) -> str:
"""Return the spec section this finding comes from."""
return BY_CODE[self.code].section
@property
def level(self) -> str:
"""Return either must or should."""
return BY_CODE[self.code].level
def __str__(self) -> str:
where = f" ({self.location})" if self.location else ""
return f"Section {self.section} — {self.message}{where}"
@dataclass
class Report:
"""What the validator found, plus the facts the directory needs."""
url: str | None = None
findings: list[Finding] = field(default_factory=list)
notes: list[str] = field(default_factory=list)
title: str = ""
description: str = ""
language: str = ""
size: int = 0
def add(self, code: str, message: str, location: str = "") -> None:
"""Record one finding, up to the report cap."""
if len(self.findings) < MAX_FINDINGS:
self.findings.append(Finding(code, message, location))
@property
def failures(self) -> list[Finding]:
"""Return the findings that make the page non-conforming."""
return [f for f in self.findings if f.level == "must"]
@property
def warnings(self) -> list[Finding]:
"""Return the findings worth fixing that still leave the page conforming."""
return [f for f in self.findings if f.level == "should"]
@property
def conforms(self) -> bool:
"""Say whether the page follows every MUST rule the validator checks."""
return not self.failures
def _entity_declarations() -> str:
"""Declare the entity set the XHTML-MP doctype would have defined.
SPEC.md 9.1 says clients must not fetch the doctype, and the local copy of
the driver DTD pulls its modules from w3.org, so the entities are supplied
from the stdlib's HTML 4 table instead — the same 252 names.
"""
return "".join(
f'<!ENTITY {name} "&#{code};">'
for code, name in sorted(html.entities.codepoint2name.items())
)
class _Resolver(etree.Resolver):
"""Resolves the XHTML-MP doctype to entity declarations and nothing else."""
def __init__(self) -> None:
self._entities = _entity_declarations()
def resolve(self, system_url, public_id, context):
"""Answer the parser's request for an external entity."""
if public_id == "-//WAPFORUM//DTD XHTML Mobile 1.2//EN":
return self.resolve_string(self._entities, context)
# With no_network set, returning None makes any other doctype fail.
return None
def _parser(*, expand_entities: bool) -> etree.XMLParser:
"""Build an XML parser that never reaches the network."""
parser = etree.XMLParser(
load_dtd=expand_entities,
resolve_entities=expand_entities,
no_network=True,
dtd_validation=False,
attribute_defaults=False,
huge_tree=False,
)
if expand_entities:
parser.resolvers.add(_Resolver())
return parser
_DTD: etree.DTD | None = None
def mews_dtd() -> etree.DTD:
"""Return the Mews DTD, loaded once per process."""
global _DTD # noqa: PLW0603
if _DTD is None:
with resources.as_file(
resources.files("mews.data").joinpath("mews-0.1.dtd")
) as path:
_DTD = etree.DTD(str(path))
return _DTD
def local(element: etree._Element) -> str:
"""Return an element's name without its namespace."""
return etree.QName(element).localname
# --- The DTD check -------------------------------------------------------
UNDECLARED_ELEMENT = re.compile(r"^No declaration for element (\S+)")
UNDECLARED_ATTRIBUTE = re.compile(
r"^No declaration for attribute (\S+) of element (\S+)"
)
NOT_ALLOWED_IN = re.compile(
r"^Element (\S+) is not declared in (\S+) list of possible children"
)
CONTENT_MODEL = re.compile(
r"^Element (\S+) content does not follow the DTD, expecting (.*?), got \(?(.*?)\)?$"
)
BAD_VALUE = re.compile(
r'^Value "(.*?)" for attribute (\S+) of (\S+) is not among the enumerated set'
)
MISSING_ATTRIBUTE = re.compile(r"^Element (\S+) does not carry attribute (\S+)")
def _enumerated_values(element: str, attribute: str) -> list[str]:
"""Return the values the DTD permits for an attribute, read from the DTD."""
for declared in mews_dtd().iterelements():
if declared.name != element:
continue
for attr in declared.iterattributes():
if attr.name == attribute:
return list(attr.itervalues() or [])
return []
def _dtd_findings(report: Report, tree: etree._ElementTree) -> None:
"""Validate against the Mews DTD and turn libxml2's errors into findings."""
dtd = mews_dtd()
if dtd.validate(tree):
return
errors = [(entry.line, entry.message) for entry in dtd.error_log]
undeclared = {
match.group(1)
for _, message in errors
for match in [UNDECLARED_ELEMENT.match(message)]
if match
}
for line, message in errors:
finding = _map_error(message, undeclared)
if finding is not None:
code, text = finding
report.add(code, text, f"line {line}")
def _map_error(message: str, undeclared: set[str]) -> tuple[str, str] | None:
"""Map one libxml2 message to a code and a plain sentence, or drop it.
One mistake makes libxml2 say several things: an unknown element is also an
unknown attribute and a content-model break. Only the clearest is kept.
"""
match = UNDECLARED_ELEMENT.match(message)
if match:
name = match.group(1)
if name == "style":
return (
"dtd-styling",
"The <style> element isn't allowed. Mews pages carry no author "
"styling, so readers control how a page looks.",
)
if name == "script":
return (
"dtd-element",
"The <script> element isn't allowed. Mews clients run no "
"scripts, so a page has to work without them.",
)
return (
"dtd-element",
f"The <{name}> element isn't allowed. Remove it, or use an element "
"from section 4.1.",
)
match = UNDECLARED_ATTRIBUTE.match(message)
if match:
attribute, element = match.group(1), match.group(2)
if element in undeclared:
return None
if attribute == "style":
return (
"dtd-styling",
f"The style attribute isn't allowed on <{element}>. Mews pages "
"carry no author styling.",
)
return (
"dtd-attribute",
f"The {attribute} attribute isn't allowed on <{element}>. Remove it.",
)
match = NOT_ALLOWED_IN.match(message)
if match:
child, parent = match.group(1), match.group(2)
if child in undeclared:
return None
if child == "table":
return (
"dtd-nested-table",
"Tables can't be nested. Move the inner table out, or use a list.",
)
return ("dtd-nesting", f"<{child}> can't go inside <{parent}>.")
match = CONTENT_MODEL.match(message)
if match:
element, got = match.group(1), match.group(3).split()
if any(name in undeclared for name in got):
return None
if element == "head":
return (
"dtd-head",
"The head can hold only a title, then meta and link elements, "
"with the title first.",
)
if not got:
if element == "body":
return ("dtd-content", "The page has nothing in its body.")
return (
"dtd-content",
f"<{element}> is empty. Either fill it or remove it.",
)
return (
"dtd-content",
f"<{element}> can't hold what's inside it here.",
)
match = BAD_VALUE.match(message)
if match:
value, attribute, element = match.groups()
if (element, attribute) == ("link", "rel"):
return (
"link-rel",
f'The link with rel="{value}" isn\'t allowed. A head may hold '
"the stylesheet link, an icon and a feed link.",
)
if (element, attribute) == ("body", "class"):
return (
"dtd-variant",
f'"{value}" isn\'t a body variant. Use mews-warm, mews-cool, '
"mews-green or mews-mono, or leave class out.",
)
allowed = ", ".join(_enumerated_values(element, attribute))
return (
"dtd-value",
f'"{value}" isn\'t allowed for {attribute} on <{element}>. '
+ (f"Use one of: {allowed}." if allowed else "Remove it."),
)
match = MISSING_ATTRIBUTE.match(message)
if match:
element, attribute = match.groups()
if element == "img" and attribute == "alt":
return (
"dtd-required-attribute",
'Every image needs an alt attribute: a description, or alt="" '
"when the image is decoration.",
)
return (
"dtd-required-attribute",
f"<{element}> needs a {attribute} attribute.",
)
return ("dtd-other", f"This page doesn't match the Mews DTD ({message}).")
# --- The rule checks -----------------------------------------------------
def _prologue(report: Report, data: bytes) -> tuple[bool, bool]:
"""Check the bytes before the root element.
Returns whether the page can be parsed at all and whether its entities may
be expanded. This runs first on purpose: a page carrying its own entity
declarations is refused before any parser sees them.
"""
head = data[:2048]
if head.startswith(b"\xef\xbb\xbf"):
report.add(
"xml-declaration",
"Remove the byte order mark at the start of the file. A Mews page "
"starts with the XML declaration.",
)
head = head[3:]
if not XML_DECLARATION.match(head):
report.add(
"xml-declaration",
'Start the page with <?xml version="1.0" encoding="UTF-8"?>.',
)
start = head.find(b"<!DOCTYPE")
if start == -1:
report.add(
"doctype",
"Add the XHTML Mobile 1.2 doctype after the XML declaration, as "
"section 3.1 shows.",
)
return True, False
end = head.find(b">", start)
if b"[" in head[start : end if end != -1 else len(head)]:
report.add(
"internal-subset",
"Remove the extra declarations from the doctype. A Mews page uses "
"the doctype from section 3.1 and nothing else.",
)
return False, False
if not DOCTYPE.match(head, start):
report.add(
"doctype",
"Use the doctype from section 3.1 exactly, pointing at the XHTML "
"Mobile 1.2 DTD.",
)
return True, False
return True, True
def _document_findings(report: Report, root: etree._Element) -> None:
"""Check the root element and the head against sections 3.1 to 3.3."""
if root.tag != f"{{{XHTML}}}html":
report.add(
"root-element",
'The root element must be <html xmlns="http://www.w3.org/1999/xhtml">.',
)
return
xml_lang, lang = root.get(XML_LANG), root.get("lang")
if not xml_lang or not lang or xml_lang != lang:
report.add(
"lang",
"Name the page language with matching xml:lang and lang attributes "
"on <html>, so browsers and clients both know it.",
)
report.language = xml_lang or lang or ""
head = root.find(f"{{{XHTML}}}head")
if head is None:
return
title = head.find(f"{{{XHTML}}}title")
if title is not None and title.text:
report.title = _clean(title.text)
markers = []
for meta in head.iterfind(f"{{{XHTML}}}meta"):
name = meta.get("name") or ""
if name not in ALLOWED_META:
report.add(
"meta-name",
f'The meta element named "{name}" isn\'t allowed. Section 3.3 '
"lists the ones a head may hold.",
)
continue
if name == MARKER:
markers.append(meta.get("content") or "")
elif name == "description":
report.description = _clean(meta.get("content") or "")
if len(markers) > 1:
report.add(
"marker",
'The head carries the <meta name="mews-profile" /> marker '
f"{len(markers)} times. Keep one.",
)
elif not markers:
report.add(
"marker",
'Add <meta name="mews-profile" content="0.1" /> to the head. '
"Clients use it to tell a Mews page apart.",
)
elif markers[0] != VERSION:
report.add(
"marker-version",
f'The marker says version "{markers[0]}". This validator checks '
f"version {VERSION}.",
)
viewport = any(
meta.get("name") == "viewport" for meta in head.iterfind(f"{{{XHTML}}}meta")
)
if not viewport:
report.add(
"viewport",
'Add <meta name="viewport" content="width=device-width" /> so phones '
"show the page at a readable size.",
)
def _link_findings(
report: Report, head: etree._Element, host: str | None
) -> tuple[str | None, bool]:
"""Check the head's links, returning the stylesheet href and if it's off-site."""
stylesheets: list[str] = []
for link in head.iterfind(f"{{{XHTML}}}link"):
rel = (link.get("rel") or "").lower()
href = link.get("href") or ""
if rel not in ALLOWED_RELS:
# Reported by the DTD check, which also names the three kinds.
continue
if rel == "stylesheet":
stylesheets.append(href)
elif rel == "alternate":
kind = (link.get("type") or "").split(";")[0].strip()
if kind != "application/atom+xml":
report.add(
"link-alternate-type",
'A feed link needs type="application/atom+xml".',
)
elif rel == "icon" and host and not _is_same_site(host, href):
report.add(
"icon-offsite",
"The icon is on another site. Host it on your own site, so "
"reading a page tells no one else about it.",
href,
)
if len(stylesheets) > 1:
report.add(
"link-stylesheet-count",
"Link the default stylesheet once. A Mews page has no other stylesheet.",
)
if not stylesheets:
report.add(
"stylesheet-missing",
"Link your copy of mews-0.1.css, so readers without a Mews client "
"still get good presentation.",
)
return None, False
href = stylesheets[0]
if host and not _is_same_site(host, href):
if _absolute(href).rstrip("/") == CANONICAL_STYLESHEET:
report.add(
"stylesheet-canonical",
"Copy mews-0.1.css to your own site and link that copy, so no "
"single server sees traffic across every Mews site.",
href,
)
else:
report.add(
"stylesheet-offsite",
"The stylesheet is on another site. A Mews page loads nothing "
"from other sites. Copy mews-0.1.css to your own site.",
href,
)
return href, True
return href, False
def _clean(text: str) -> str:
"""Collapse whitespace and drop characters that could reshape a listing."""
# Bidirectional overrides are printable but can make a title read as a
# different domain, so they go too.
overrides = frozenset(
chr(code) for code in (*range(0x202A, 0x202F), *range(0x2066, 0x206A))
)
stripped = "".join(
character
for character in text
if character.isprintable() and character not in overrides
)
return " ".join(stripped.split())
def _is_same_site(host: str, href: str) -> bool:
"""Say whether an href stays on the page's registered domain."""
other = urlsplit(href).hostname
return other is None or same_site(host, other)
def _absolute(href: str) -> str:
"""Treat an href with no scheme as https, for comparing with a known URL."""
return href if "://" in href else "https://" + href.lstrip("/")
def _transport_findings(report: Report, response: Fetched) -> None:
"""Check the response headers against sections 7.1 to 7.4."""
if "set-cookie" in response.headers:
report.add(
"cookie",
"The server sets a cookie on this page. A Mews page sets cookies "
"only for a form the reader submits.",
)
kind = (response.headers.get("content-type") or "").split(";")[0].strip().lower()
if kind not in (
"text/html",
"application/xhtml+xml",
"application/vnd.wap.xhtml+xml",
):
report.add(
"content-type",
f'The server sends this page as "{kind or "nothing"}". Send it as '
"text/html, so a browser still shows a page with a small error.",
)
if not response.headers.get("last-modified") and not response.headers.get("etag"):
report.add(
"validators",
"Send a Last-Modified or ETag header, so clients and the directory "
"can check for changes cheaply.",
)
if urlsplit(response.url).scheme != "https" or response.scheme_downgraded:
report.add(
"https",
"Serve the page over HTTPS. You may serve plain HTTP in parallel "
"for old handsets.",
)
def _image_findings(
report: Report, root: etree._Element, url: str | None, fetcher: Fetcher | None
) -> int:
"""Check every image, returning the bytes the fetched ones took."""
host = urlsplit(url).hostname if url else None
images = list(root.iter(f"{{{XHTML}}}img"))
total = 0
fetched = 0
for image in images:
src = (image.get("src") or "").strip()
if src.lower().startswith("data:"):
report.add(
"image-data-uri",
"This image is built into the page as a data: URI. Save it as a "
"file on your own site and link it.",
)
continue
if not src:
continue
target = urljoin(url, src) if url else src
if host and not _is_same_site(host, target):
report.add(
"image-offsite",
"This image is on another site. Host it on your own site, so "
"reading a page tells no one else about it.",
src,
)
continue
if fetcher is None or url is None:
continue
if fetched >= MAX_IMAGES:
continue
fetched += 1
total += _check_image(report, target, src, fetcher)
if fetcher is not None and len(images) > MAX_IMAGES:
report.add(
"images-not-checked",
f"This page has {len(images)} images and the checker read the first "
f"{MAX_IMAGES}. Check the rest yourself.",
)
return total
def _check_image(report: Report, target: str, src: str, fetcher: Fetcher) -> int:
"""Fetch one image and check its format, size and metadata."""
try:
response = fetcher.get(target, cap=IMAGE_CAP)
except (UrlError, FetchError) as error:
report.add("image-unreachable", f"This image didn't load: {error}", src)
return 0
if response.status != OK:
report.add(
"image-unreachable",
f"This image didn't load (it returned {response.status}).",
src,
)
return 0
if "set-cookie" in response.headers:
report.add("cookie", "The server sets a cookie on this image.", src)
body = response.body
if response.truncated:
report.add(
"image-size",
f"This image is over {IMAGE_CAP // 1024} KB. Keep images under "
"50 KB, so a page loads quickly on a slow connection.",
src,
)
elif len(body) > IMAGE_SHOULD:
report.add(
"image-size",
f"This image is {len(body) // 1024} KB. Keep images under 50 KB, so "
"a page loads quickly on a slow connection.",
src,
)
if not any(body.startswith(magic) for magic in IMAGE_MAGIC):
report.add(
"image-format",
"This image isn't a GIF, JPEG or PNG. Use one of those, so every "
"client and old handset can show it.",
src,
)
elif _has_metadata(body):
report.add(
"image-metadata",
"This image carries camera or location metadata. Strip it before "
"publishing.",
src,
)
return len(body)
def _has_metadata(body: bytes) -> bool:
"""Say whether an image carries an Exif block."""
if body.startswith(b"\xff\xd8\xff"):
return b"Exif\x00\x00" in body[:4096]
if body.startswith(b"\x89PNG"):
return b"eXIf" in body[:4096]
return False
def _stylesheet_findings(report: Report, href: str, url: str, fetcher: Fetcher) -> None:
"""Fetch the linked stylesheet and compare it with the default one."""
target = urljoin(url, href)
host = urlsplit(url).hostname or ""
try:
response = fetcher.get(target, cap=CSS_CAP)
except (UrlError, FetchError) as error:
report.add(
"stylesheet-unreachable", f"The stylesheet didn't load: {error}", href
)
return
if response.status != OK:
report.add(
"stylesheet-unreachable",
f"The stylesheet didn't load (it returned {response.status}).",
href,
)
return
if "set-cookie" in response.headers:
report.add("cookie", "The server sets a cookie on the stylesheet.", href)
comparison = mews_css.compare(response.body.decode("utf-8", "replace"))
if comparison.has_import:
report.add(
"stylesheet-import",
"Your stylesheet copy uses @import. Remove it: a Mews page loads "
"one stylesheet and nothing else.",
href,
)
if not comparison.equal:
report.add(
"stylesheet-modified",
"Your copy of mews-0.1.css has been changed, starting around line "
f"{comparison.diff_line}. Copy it again unchanged. You may add "
"@font-face rules that load fonts from your own site.",
href,
)
for font in comparison.font_urls:
if not _is_same_site(host, urljoin(target, font)):
report.add(
"font-offsite",
"An @font-face rule loads a font from another site. Host the "
"font file on your own site.",
font,
)
# --- Entry points --------------------------------------------------------
def validate_bytes(
data: bytes,
*,
url: str | None = None,
response: Fetched | None = None,
fetcher: Fetcher | None = None,
) -> Report:
"""Validate one page.
url is the address the page was fetched from, which the same-site rules in
4.2 and 7.4 need. Without a fetcher the checks that need the live page are
skipped and said to be skipped.
"""
report = Report(url=url)
report.size = len(data)
can_parse, expand_entities = _prologue(report, data)
if not can_parse:
return report
try:
root = etree.fromstring(data, _parser(expand_entities=expand_entities))
except etree.XMLSyntaxError as error:
report.add(
"well-formed",
"This page isn't well-formed XML, so a Mews client can't read it "
f"({error.msg}).",
f"line {error.lineno}",
)
return report
tree = etree.ElementTree(root)
_document_findings(report, root)
_dtd_findings(report, tree)
host = urlsplit(url).hostname if url else None
head = root.find(f"{{{XHTML}}}head")
stylesheet, offsite = (None, False)
if head is not None:
stylesheet, offsite = _link_findings(report, head, host)
if response is not None:
_transport_findings(report, response)
image_bytes = _image_findings(report, root, url, fetcher)
if stylesheet and not offsite and url and fetcher is not None:
_stylesheet_findings(report, stylesheet, url, fetcher)
if (response is not None and response.truncated) or report.size > SIZE_MUST:
report.add(
"size-must",
f"The page markup is over {SIZE_MUST // 1024} KB. Split it into "
"several pages.",
)
elif report.size > SIZE_SHOULD:
report.add(
"size-should",
f"The page markup is {report.size // 1024} KB. Keep it under "
"64 KB, so it loads quickly on a slow connection.",
)
if fetcher is not None and report.size + image_bytes > SIZE_TOTAL:
report.add(
"size-total",
"The page and its images come to more than 320 KB together. Use "
"fewer or smaller images.",
)
if fetcher is None:
report.notes.append(
"The checks that need the live page were skipped: response headers "
"(7.1 to 7.4), image sizes and formats (4.2), and the stylesheet "
"copy (5.2)."
)
return report
def validate_file(path: str) -> Report:
"""Validate a page on disk."""
with open(path, "rb") as handle:
return validate_bytes(handle.read())
def validate_url(
url: str, *, fetcher: Fetcher | None = None, allow_loopback: bool = False
) -> Report:
"""Fetch a page and validate it, along with its stylesheet and images."""
owned = fetcher is None
client = fetcher or Fetcher(allow_loopback=allow_loopback)
try:
response = client.get(normalise_url(url))
if response.status != OK:
report = Report(url=response.url)
report.add(
"well-formed",
f"We couldn't read that page (it returned {response.status}). "
"Check the address and try again.",
)
return report
return validate_bytes(
response.body, url=response.url, response=response, fetcher=client
)
finally:
if owned:
client.close()
def main(argv: list[str] | None = None) -> int:
"""Check pages named on the command line, returning 1 if any doesn't conform."""
parser = argparse.ArgumentParser(
description="Check a page against Mews Profile 0.1 (see doc/SPEC.md)."
)
parser.add_argument("target", nargs="+", help="a page address or a file")
parser.add_argument(
"--local",
action="store_true",
help="also read addresses on this machine, for checking a site before "
"it is published",
)
parser.add_argument(
"--quiet", action="store_true", help="print nothing for pages that conform"
)
args = parser.parse_args(argv)
worst = 0
for target in args.target:
if "://" in target or not _looks_like_path(target):
try:
report = validate_url(target, allow_loopback=args.local)
except (UrlError, FetchError) as error:
print(f"{target} — {error}", file=sys.stderr)
worst = 1
continue
else:
report = validate_file(target)
if report.conforms and args.quiet:
continue
worst = max(worst, 0 if report.conforms else 1)
verdict = "conforms" if report.conforms else "doesn't conform"
print(f"{target} — {verdict}")
for finding in report.failures:
print(f" {finding}")
if report.warnings:
print(" Worth fixing:")
for finding in report.warnings:
print(f" {finding}")
for note in report.notes:
print(f" Note: {note}")
return worst
def _looks_like_path(target: str) -> bool:
"""Say whether a bare argument names a file rather than a site."""
return Path(target).exists()