976 lines
34 KiB
Python
976 lines
34 KiB
Python
"""The Mews page validator.
|
|
|
|
SPEC.md 9.1 defines conformance as two checks: the page validates against
|
|
dtd/mews-0.1.dtd, and it follows the rules a DTD cannot express. This module
|
|
runs both and reports every failure with its spec section.
|
|
|
|
Failures at MUST level mean the page does not conform. Failures at SHOULD level
|
|
are warnings: they are worth fixing but do not make a page non-conforming.
|
|
"""
|
|
|
|
import argparse
|
|
from dataclasses import dataclass, field
|
|
import html.entities
|
|
from importlib import resources
|
|
from pathlib import Path
|
|
import re
|
|
import sys
|
|
from urllib.parse import urljoin, urlsplit
|
|
|
|
from lxml import etree
|
|
|
|
from mews import VERSION, css as mews_css
|
|
from mews.fetch import (
|
|
CSS_CAP,
|
|
IMAGE_CAP,
|
|
Fetched,
|
|
Fetcher,
|
|
FetchError,
|
|
UrlError,
|
|
normalise_url,
|
|
same_site,
|
|
)
|
|
|
|
XHTML = "http://www.w3.org/1999/xhtml"
|
|
XML_LANG = "{http://www.w3.org/XML/1998/namespace}lang"
|
|
|
|
MARKER = "mews-profile"
|
|
CANONICAL_STYLESHEET = "https://mews.page/mews-0.1.css"
|
|
|
|
OK = 200
|
|
MAX_FINDINGS = 50
|
|
MAX_IMAGES = 20
|
|
|
|
SIZE_MUST = 256 * 1024
|
|
SIZE_SHOULD = 64 * 1024
|
|
SIZE_TOTAL = 320 * 1024
|
|
IMAGE_SHOULD = 50 * 1024
|
|
|
|
XML_DECLARATION = re.compile(rb'^<\?xml version="1\.0" encoding="(?i:UTF-8)"\?>')
|
|
DOCTYPE = re.compile(
|
|
rb'<!DOCTYPE\s+html\s+PUBLIC\s+"-//WAPFORUM//DTD XHTML Mobile 1\.2//EN"\s+'
|
|
rb'"http://www\.openmobilealliance\.org/tech/DTD/xhtml-mobile12\.dtd"\s*>'
|
|
)
|
|
|
|
ALLOWED_META = ("mews-profile", "description", "author", "viewport")
|
|
ALLOWED_RELS = ("stylesheet", "icon", "alternate")
|
|
|
|
# Magic bytes for the three formats SPEC.md 4.2 recommends.
|
|
IMAGE_MAGIC = {
|
|
b"GIF87a": "GIF",
|
|
b"GIF89a": "GIF",
|
|
b"\x89PNG\r\n\x1a\n": "PNG",
|
|
b"\xff\xd8\xff": "JPEG",
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Check:
|
|
"""One thing the validator looks at, and the spec section it comes from."""
|
|
|
|
code: str
|
|
section: str
|
|
level: str
|
|
what: str
|
|
|
|
|
|
CHECKS: tuple[Check, ...] = (
|
|
Check("xml-declaration", "3.1", "must", "UTF-8 XML declaration, no BOM"),
|
|
Check("doctype", "3.1", "must", "the XHTML-MP 1.2 doctype, exactly"),
|
|
Check("internal-subset", "3.1", "must", "no internal DTD subset"),
|
|
Check("well-formed", "3", "must", "the page parses as XML"),
|
|
Check("root-element", "3.1", "must", "html root in the XHTML namespace"),
|
|
Check("lang", "3.1", "should", "xml:lang and lang agree on html"),
|
|
Check("dtd-element", "4.1", "must", "only elements from 4.1"),
|
|
Check("dtd-attribute", "4.1", "must", "only attributes from 4.1"),
|
|
Check("dtd-nesting", "4.1", "must", "elements nest as the DTD allows"),
|
|
Check("dtd-content", "4.1", "must", "element content follows the DTD"),
|
|
Check("dtd-value", "4.1", "must", "attribute values from the permitted set"),
|
|
Check("dtd-required-attribute", "4.1", "must", "required attributes present"),
|
|
Check("dtd-head", "3.3", "must", "the head holds only what 3.3 permits"),
|
|
Check("dtd-nested-table", "4.2", "must", "tables are not nested"),
|
|
Check("dtd-variant", "5.3", "must", "body class is one of the variants"),
|
|
Check("dtd-styling", "5", "must", "no author styling"),
|
|
Check("dtd-other", "4.1", "must", "the page matches the Mews DTD"),
|
|
Check("marker", "3.2", "must", "the conformance marker is present once"),
|
|
Check("marker-version", "3.2", "must", "the marker names a known version"),
|
|
Check("meta-name", "3.3", "must", "only the meta names 3.3 permits"),
|
|
Check("link-rel", "3.3", "must", "only the link kinds 3.3 permits"),
|
|
Check("link-stylesheet-count", "3.3", "must", "at most one stylesheet link"),
|
|
Check("link-alternate-type", "3.3", "must", "feed links are Atom"),
|
|
Check("icon-offsite", "3.3", "must", "the icon is on the page's own site"),
|
|
Check("viewport", "3.3", "should", "the viewport meta is present"),
|
|
Check("size-must", "4.3", "must", "markup under 256 KB"),
|
|
Check("size-should", "4.3", "should", "markup under 64 KB"),
|
|
Check("size-total", "4.3", "should", "page plus images under 320 KB"),
|
|
Check("image-data-uri", "4.2", "must", "no data: URI images"),
|
|
Check("image-offsite", "4.2", "must", "images are on the page's own site"),
|
|
Check("image-format", "4.2", "should", "images are GIF, JPEG or PNG"),
|
|
Check("image-size", "4.2", "should", "each image under 50 KB"),
|
|
Check("image-unreachable", "4.2", "should", "images load"),
|
|
Check("image-metadata", "4.2", "should", "images carry no camera metadata"),
|
|
Check("images-not-checked", "4.2", "should", "how many images were checked"),
|
|
Check("stylesheet-missing", "5.2", "should", "the default stylesheet is linked"),
|
|
Check("stylesheet-canonical", "5.2", "should", "the site hosts its own copy"),
|
|
Check("stylesheet-modified", "5.2", "must", "the copy is unmodified"),
|
|
Check("stylesheet-import", "5.2", "must", "the copy has no @import"),
|
|
Check("stylesheet-unreachable", "5.2", "should", "the stylesheet loads"),
|
|
Check("font-offsite", "5.2", "must", "@font-face fonts are on the own site"),
|
|
Check("stylesheet-offsite", "7.4", "must", "the stylesheet is on the own site"),
|
|
Check("content-type", "7.1", "should", "a page content type"),
|
|
Check("https", "7.2", "should", "served over HTTPS"),
|
|
Check("validators", "7.3", "should", "Last-Modified or ETag is sent"),
|
|
Check("cookie", "7.4", "must", "no cookies are set"),
|
|
)
|
|
|
|
BY_CODE = {check.code: check for check in CHECKS}
|
|
|
|
# Rules in sections 3 to 7 this validator deliberately leaves alone. Together
|
|
# with CHECKS this covers every MUST and SHOULD in those sections; tests fail if
|
|
# a section appears in neither.
|
|
NOT_CHECKED = {
|
|
"5.1": "The default stylesheet is compared byte for byte (5.2), which "
|
|
"covers whether its base rules are valid WAP CSS.",
|
|
"6.1": "Dated links are a client convention; the validator reads no feeds.",
|
|
"6.2": "Atom feed contents are a client convention; only the link type is checked.",
|
|
"7.2": "Whether a site redirects HTTP to HTTPS in a way old handsets can "
|
|
"follow cannot be told from one request.",
|
|
"4.2": "Whether text appears only inside an image cannot be told from markup.",
|
|
}
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Finding:
|
|
"""One rule a page broke."""
|
|
|
|
code: str
|
|
message: str
|
|
location: str = ""
|
|
|
|
@property
|
|
def section(self) -> str:
|
|
"""Return the spec section this finding comes from."""
|
|
return BY_CODE[self.code].section
|
|
|
|
@property
|
|
def level(self) -> str:
|
|
"""Return either must or should."""
|
|
return BY_CODE[self.code].level
|
|
|
|
def __str__(self) -> str:
|
|
where = f" ({self.location})" if self.location else ""
|
|
return f"Section {self.section} — {self.message}{where}"
|
|
|
|
|
|
@dataclass
|
|
class Report:
|
|
"""What the validator found, plus the facts the directory needs."""
|
|
|
|
url: str | None = None
|
|
findings: list[Finding] = field(default_factory=list)
|
|
notes: list[str] = field(default_factory=list)
|
|
title: str = ""
|
|
description: str = ""
|
|
language: str = ""
|
|
size: int = 0
|
|
|
|
def add(self, code: str, message: str, location: str = "") -> None:
|
|
"""Record one finding, up to the report cap."""
|
|
if len(self.findings) < MAX_FINDINGS:
|
|
self.findings.append(Finding(code, message, location))
|
|
|
|
@property
|
|
def failures(self) -> list[Finding]:
|
|
"""Return the findings that make the page non-conforming."""
|
|
return [f for f in self.findings if f.level == "must"]
|
|
|
|
@property
|
|
def warnings(self) -> list[Finding]:
|
|
"""Return the findings worth fixing that still leave the page conforming."""
|
|
return [f for f in self.findings if f.level == "should"]
|
|
|
|
@property
|
|
def conforms(self) -> bool:
|
|
"""Say whether the page follows every MUST rule the validator checks."""
|
|
return not self.failures
|
|
|
|
|
|
def _entity_declarations() -> str:
|
|
"""Declare the entity set the XHTML-MP doctype would have defined.
|
|
|
|
SPEC.md 9.1 says clients must not fetch the doctype, and the local copy of
|
|
the driver DTD pulls its modules from w3.org, so the entities are supplied
|
|
from the stdlib's HTML 4 table instead — the same 252 names.
|
|
"""
|
|
return "".join(
|
|
f'<!ENTITY {name} "&#{code};">'
|
|
for code, name in sorted(html.entities.codepoint2name.items())
|
|
)
|
|
|
|
|
|
class _Resolver(etree.Resolver):
|
|
"""Resolves the XHTML-MP doctype to entity declarations and nothing else."""
|
|
|
|
def __init__(self) -> None:
|
|
self._entities = _entity_declarations()
|
|
|
|
def resolve(self, system_url, public_id, context):
|
|
"""Answer the parser's request for an external entity."""
|
|
if public_id == "-//WAPFORUM//DTD XHTML Mobile 1.2//EN":
|
|
return self.resolve_string(self._entities, context)
|
|
# With no_network set, returning None makes any other doctype fail.
|
|
return None
|
|
|
|
|
|
def _parser(*, expand_entities: bool) -> etree.XMLParser:
|
|
"""Build an XML parser that never reaches the network."""
|
|
parser = etree.XMLParser(
|
|
load_dtd=expand_entities,
|
|
resolve_entities=expand_entities,
|
|
no_network=True,
|
|
dtd_validation=False,
|
|
attribute_defaults=False,
|
|
huge_tree=False,
|
|
)
|
|
if expand_entities:
|
|
parser.resolvers.add(_Resolver())
|
|
return parser
|
|
|
|
|
|
_DTD: etree.DTD | None = None
|
|
|
|
|
|
def mews_dtd() -> etree.DTD:
|
|
"""Return the Mews DTD, loaded once per process."""
|
|
global _DTD # noqa: PLW0603
|
|
if _DTD is None:
|
|
with resources.as_file(
|
|
resources.files("mews.data").joinpath("mews-0.1.dtd")
|
|
) as path:
|
|
_DTD = etree.DTD(str(path))
|
|
return _DTD
|
|
|
|
|
|
def local(element: etree._Element) -> str:
|
|
"""Return an element's name without its namespace."""
|
|
return etree.QName(element).localname
|
|
|
|
|
|
# --- The DTD check -------------------------------------------------------
|
|
|
|
UNDECLARED_ELEMENT = re.compile(r"^No declaration for element (\S+)")
|
|
UNDECLARED_ATTRIBUTE = re.compile(
|
|
r"^No declaration for attribute (\S+) of element (\S+)"
|
|
)
|
|
NOT_ALLOWED_IN = re.compile(
|
|
r"^Element (\S+) is not declared in (\S+) list of possible children"
|
|
)
|
|
CONTENT_MODEL = re.compile(
|
|
r"^Element (\S+) content does not follow the DTD, expecting (.*?), got \(?(.*?)\)?$"
|
|
)
|
|
BAD_VALUE = re.compile(
|
|
r'^Value "(.*?)" for attribute (\S+) of (\S+) is not among the enumerated set'
|
|
)
|
|
MISSING_ATTRIBUTE = re.compile(r"^Element (\S+) does not carry attribute (\S+)")
|
|
|
|
|
|
def _enumerated_values(element: str, attribute: str) -> list[str]:
|
|
"""Return the values the DTD permits for an attribute, read from the DTD."""
|
|
for declared in mews_dtd().iterelements():
|
|
if declared.name != element:
|
|
continue
|
|
for attr in declared.iterattributes():
|
|
if attr.name == attribute:
|
|
return list(attr.itervalues() or [])
|
|
return []
|
|
|
|
|
|
def _dtd_findings(report: Report, tree: etree._ElementTree) -> None:
|
|
"""Validate against the Mews DTD and turn libxml2's errors into findings."""
|
|
dtd = mews_dtd()
|
|
if dtd.validate(tree):
|
|
return
|
|
|
|
errors = [(entry.line, entry.message) for entry in dtd.error_log]
|
|
undeclared = {
|
|
match.group(1)
|
|
for _, message in errors
|
|
for match in [UNDECLARED_ELEMENT.match(message)]
|
|
if match
|
|
}
|
|
|
|
for line, message in errors:
|
|
finding = _map_error(message, undeclared)
|
|
if finding is not None:
|
|
code, text = finding
|
|
report.add(code, text, f"line {line}")
|
|
|
|
|
|
def _map_error(message: str, undeclared: set[str]) -> tuple[str, str] | None:
|
|
"""Map one libxml2 message to a code and a plain sentence, or drop it.
|
|
|
|
One mistake makes libxml2 say several things: an unknown element is also an
|
|
unknown attribute and a content-model break. Only the clearest is kept.
|
|
"""
|
|
match = UNDECLARED_ELEMENT.match(message)
|
|
if match:
|
|
name = match.group(1)
|
|
if name == "style":
|
|
return (
|
|
"dtd-styling",
|
|
"The <style> element isn't allowed. Mews pages carry no author "
|
|
"styling, so readers control how a page looks.",
|
|
)
|
|
if name == "script":
|
|
return (
|
|
"dtd-element",
|
|
"The <script> element isn't allowed. Mews clients run no "
|
|
"scripts, so a page has to work without them.",
|
|
)
|
|
return (
|
|
"dtd-element",
|
|
f"The <{name}> element isn't allowed. Remove it, or use an element "
|
|
"from section 4.1.",
|
|
)
|
|
|
|
match = UNDECLARED_ATTRIBUTE.match(message)
|
|
if match:
|
|
attribute, element = match.group(1), match.group(2)
|
|
if element in undeclared:
|
|
return None
|
|
if attribute == "style":
|
|
return (
|
|
"dtd-styling",
|
|
f"The style attribute isn't allowed on <{element}>. Mews pages "
|
|
"carry no author styling.",
|
|
)
|
|
return (
|
|
"dtd-attribute",
|
|
f"The {attribute} attribute isn't allowed on <{element}>. Remove it.",
|
|
)
|
|
|
|
match = NOT_ALLOWED_IN.match(message)
|
|
if match:
|
|
child, parent = match.group(1), match.group(2)
|
|
if child in undeclared:
|
|
return None
|
|
if child == "table":
|
|
return (
|
|
"dtd-nested-table",
|
|
"Tables can't be nested. Move the inner table out, or use a list.",
|
|
)
|
|
return ("dtd-nesting", f"<{child}> can't go inside <{parent}>.")
|
|
|
|
match = CONTENT_MODEL.match(message)
|
|
if match:
|
|
element, got = match.group(1), match.group(3).split()
|
|
if any(name in undeclared for name in got):
|
|
return None
|
|
if element == "head":
|
|
return (
|
|
"dtd-head",
|
|
"The head can hold only a title, then meta and link elements, "
|
|
"with the title first.",
|
|
)
|
|
if not got:
|
|
if element == "body":
|
|
return ("dtd-content", "The page has nothing in its body.")
|
|
return (
|
|
"dtd-content",
|
|
f"<{element}> is empty. Either fill it or remove it.",
|
|
)
|
|
return (
|
|
"dtd-content",
|
|
f"<{element}> can't hold what's inside it here.",
|
|
)
|
|
|
|
match = BAD_VALUE.match(message)
|
|
if match:
|
|
value, attribute, element = match.groups()
|
|
if (element, attribute) == ("link", "rel"):
|
|
return (
|
|
"link-rel",
|
|
f'The link with rel="{value}" isn\'t allowed. A head may hold '
|
|
"the stylesheet link, an icon and a feed link.",
|
|
)
|
|
if (element, attribute) == ("body", "class"):
|
|
return (
|
|
"dtd-variant",
|
|
f'"{value}" isn\'t a body variant. Use mews-warm, mews-cool, '
|
|
"mews-green or mews-mono, or leave class out.",
|
|
)
|
|
allowed = ", ".join(_enumerated_values(element, attribute))
|
|
return (
|
|
"dtd-value",
|
|
f'"{value}" isn\'t allowed for {attribute} on <{element}>. '
|
|
+ (f"Use one of: {allowed}." if allowed else "Remove it."),
|
|
)
|
|
|
|
match = MISSING_ATTRIBUTE.match(message)
|
|
if match:
|
|
element, attribute = match.groups()
|
|
if element == "img" and attribute == "alt":
|
|
return (
|
|
"dtd-required-attribute",
|
|
'Every image needs an alt attribute: a description, or alt="" '
|
|
"when the image is decoration.",
|
|
)
|
|
return (
|
|
"dtd-required-attribute",
|
|
f"<{element}> needs a {attribute} attribute.",
|
|
)
|
|
|
|
return ("dtd-other", f"This page doesn't match the Mews DTD ({message}).")
|
|
|
|
|
|
# --- The rule checks -----------------------------------------------------
|
|
|
|
|
|
def _prologue(report: Report, data: bytes) -> tuple[bool, bool]:
|
|
"""Check the bytes before the root element.
|
|
|
|
Returns whether the page can be parsed at all and whether its entities may
|
|
be expanded. This runs first on purpose: a page carrying its own entity
|
|
declarations is refused before any parser sees them.
|
|
"""
|
|
head = data[:2048]
|
|
if head.startswith(b"\xef\xbb\xbf"):
|
|
report.add(
|
|
"xml-declaration",
|
|
"Remove the byte order mark at the start of the file. A Mews page "
|
|
"starts with the XML declaration.",
|
|
)
|
|
head = head[3:]
|
|
if not XML_DECLARATION.match(head):
|
|
report.add(
|
|
"xml-declaration",
|
|
'Start the page with <?xml version="1.0" encoding="UTF-8"?>.',
|
|
)
|
|
|
|
start = head.find(b"<!DOCTYPE")
|
|
if start == -1:
|
|
report.add(
|
|
"doctype",
|
|
"Add the XHTML Mobile 1.2 doctype after the XML declaration, as "
|
|
"section 3.1 shows.",
|
|
)
|
|
return True, False
|
|
|
|
end = head.find(b">", start)
|
|
if b"[" in head[start : end if end != -1 else len(head)]:
|
|
report.add(
|
|
"internal-subset",
|
|
"Remove the extra declarations from the doctype. A Mews page uses "
|
|
"the doctype from section 3.1 and nothing else.",
|
|
)
|
|
return False, False
|
|
|
|
if not DOCTYPE.match(head, start):
|
|
report.add(
|
|
"doctype",
|
|
"Use the doctype from section 3.1 exactly, pointing at the XHTML "
|
|
"Mobile 1.2 DTD.",
|
|
)
|
|
return True, False
|
|
return True, True
|
|
|
|
|
|
def _document_findings(report: Report, root: etree._Element) -> None:
|
|
"""Check the root element and the head against sections 3.1 to 3.3."""
|
|
if root.tag != f"{{{XHTML}}}html":
|
|
report.add(
|
|
"root-element",
|
|
'The root element must be <html xmlns="http://www.w3.org/1999/xhtml">.',
|
|
)
|
|
return
|
|
|
|
xml_lang, lang = root.get(XML_LANG), root.get("lang")
|
|
if not xml_lang or not lang or xml_lang != lang:
|
|
report.add(
|
|
"lang",
|
|
"Name the page language with matching xml:lang and lang attributes "
|
|
"on <html>, so browsers and clients both know it.",
|
|
)
|
|
report.language = xml_lang or lang or ""
|
|
|
|
head = root.find(f"{{{XHTML}}}head")
|
|
if head is None:
|
|
return
|
|
|
|
title = head.find(f"{{{XHTML}}}title")
|
|
if title is not None and title.text:
|
|
report.title = _clean(title.text)
|
|
|
|
markers = []
|
|
for meta in head.iterfind(f"{{{XHTML}}}meta"):
|
|
name = meta.get("name") or ""
|
|
if name not in ALLOWED_META:
|
|
report.add(
|
|
"meta-name",
|
|
f'The meta element named "{name}" isn\'t allowed. Section 3.3 '
|
|
"lists the ones a head may hold.",
|
|
)
|
|
continue
|
|
if name == MARKER:
|
|
markers.append(meta.get("content") or "")
|
|
elif name == "description":
|
|
report.description = _clean(meta.get("content") or "")
|
|
|
|
if len(markers) > 1:
|
|
report.add(
|
|
"marker",
|
|
'The head carries the <meta name="mews-profile" /> marker '
|
|
f"{len(markers)} times. Keep one.",
|
|
)
|
|
elif not markers:
|
|
report.add(
|
|
"marker",
|
|
'Add <meta name="mews-profile" content="0.1" /> to the head. '
|
|
"Clients use it to tell a Mews page apart.",
|
|
)
|
|
elif markers[0] != VERSION:
|
|
report.add(
|
|
"marker-version",
|
|
f'The marker says version "{markers[0]}". This validator checks '
|
|
f"version {VERSION}.",
|
|
)
|
|
|
|
viewport = any(
|
|
meta.get("name") == "viewport" for meta in head.iterfind(f"{{{XHTML}}}meta")
|
|
)
|
|
if not viewport:
|
|
report.add(
|
|
"viewport",
|
|
'Add <meta name="viewport" content="width=device-width" /> so phones '
|
|
"show the page at a readable size.",
|
|
)
|
|
|
|
|
|
def _link_findings(
|
|
report: Report, head: etree._Element, host: str | None
|
|
) -> tuple[str | None, bool]:
|
|
"""Check the head's links, returning the stylesheet href and if it's off-site."""
|
|
stylesheets: list[str] = []
|
|
for link in head.iterfind(f"{{{XHTML}}}link"):
|
|
rel = (link.get("rel") or "").lower()
|
|
href = link.get("href") or ""
|
|
if rel not in ALLOWED_RELS:
|
|
# Reported by the DTD check, which also names the three kinds.
|
|
continue
|
|
if rel == "stylesheet":
|
|
stylesheets.append(href)
|
|
elif rel == "alternate":
|
|
kind = (link.get("type") or "").split(";")[0].strip()
|
|
if kind != "application/atom+xml":
|
|
report.add(
|
|
"link-alternate-type",
|
|
'A feed link needs type="application/atom+xml".',
|
|
)
|
|
elif rel == "icon" and host and not _is_same_site(host, href):
|
|
report.add(
|
|
"icon-offsite",
|
|
"The icon is on another site. Host it on your own site, so "
|
|
"reading a page tells no one else about it.",
|
|
href,
|
|
)
|
|
|
|
if len(stylesheets) > 1:
|
|
report.add(
|
|
"link-stylesheet-count",
|
|
"Link the default stylesheet once. A Mews page has no other stylesheet.",
|
|
)
|
|
if not stylesheets:
|
|
report.add(
|
|
"stylesheet-missing",
|
|
"Link your copy of mews-0.1.css, so readers without a Mews client "
|
|
"still get good presentation.",
|
|
)
|
|
return None, False
|
|
|
|
href = stylesheets[0]
|
|
if host and not _is_same_site(host, href):
|
|
if _absolute(href).rstrip("/") == CANONICAL_STYLESHEET:
|
|
report.add(
|
|
"stylesheet-canonical",
|
|
"Copy mews-0.1.css to your own site and link that copy, so no "
|
|
"single server sees traffic across every Mews site.",
|
|
href,
|
|
)
|
|
else:
|
|
report.add(
|
|
"stylesheet-offsite",
|
|
"The stylesheet is on another site. A Mews page loads nothing "
|
|
"from other sites. Copy mews-0.1.css to your own site.",
|
|
href,
|
|
)
|
|
return href, True
|
|
return href, False
|
|
|
|
|
|
def _clean(text: str) -> str:
|
|
"""Collapse whitespace and drop characters that could reshape a listing."""
|
|
# Bidirectional overrides are printable but can make a title read as a
|
|
# different domain, so they go too.
|
|
overrides = frozenset(
|
|
chr(code) for code in (*range(0x202A, 0x202F), *range(0x2066, 0x206A))
|
|
)
|
|
stripped = "".join(
|
|
character
|
|
for character in text
|
|
if character.isprintable() and character not in overrides
|
|
)
|
|
return " ".join(stripped.split())
|
|
|
|
|
|
def _is_same_site(host: str, href: str) -> bool:
|
|
"""Say whether an href stays on the page's registered domain."""
|
|
other = urlsplit(href).hostname
|
|
return other is None or same_site(host, other)
|
|
|
|
|
|
def _absolute(href: str) -> str:
|
|
"""Treat an href with no scheme as https, for comparing with a known URL."""
|
|
return href if "://" in href else "https://" + href.lstrip("/")
|
|
|
|
|
|
def _transport_findings(report: Report, response: Fetched) -> None:
|
|
"""Check the response headers against sections 7.1 to 7.4."""
|
|
if "set-cookie" in response.headers:
|
|
report.add(
|
|
"cookie",
|
|
"The server sets a cookie on this page. A Mews page sets cookies "
|
|
"only for a form the reader submits.",
|
|
)
|
|
kind = (response.headers.get("content-type") or "").split(";")[0].strip().lower()
|
|
if kind not in (
|
|
"text/html",
|
|
"application/xhtml+xml",
|
|
"application/vnd.wap.xhtml+xml",
|
|
):
|
|
report.add(
|
|
"content-type",
|
|
f'The server sends this page as "{kind or "nothing"}". Send it as '
|
|
"text/html, so a browser still shows a page with a small error.",
|
|
)
|
|
if not response.headers.get("last-modified") and not response.headers.get("etag"):
|
|
report.add(
|
|
"validators",
|
|
"Send a Last-Modified or ETag header, so clients and the directory "
|
|
"can check for changes cheaply.",
|
|
)
|
|
if urlsplit(response.url).scheme != "https" or response.scheme_downgraded:
|
|
report.add(
|
|
"https",
|
|
"Serve the page over HTTPS. You may serve plain HTTP in parallel "
|
|
"for old handsets.",
|
|
)
|
|
|
|
|
|
def _image_findings(
|
|
report: Report, root: etree._Element, url: str | None, fetcher: Fetcher | None
|
|
) -> int:
|
|
"""Check every image, returning the bytes the fetched ones took."""
|
|
host = urlsplit(url).hostname if url else None
|
|
images = list(root.iter(f"{{{XHTML}}}img"))
|
|
total = 0
|
|
fetched = 0
|
|
for image in images:
|
|
src = (image.get("src") or "").strip()
|
|
if src.lower().startswith("data:"):
|
|
report.add(
|
|
"image-data-uri",
|
|
"This image is built into the page as a data: URI. Save it as a "
|
|
"file on your own site and link it.",
|
|
)
|
|
continue
|
|
if not src:
|
|
continue
|
|
target = urljoin(url, src) if url else src
|
|
if host and not _is_same_site(host, target):
|
|
report.add(
|
|
"image-offsite",
|
|
"This image is on another site. Host it on your own site, so "
|
|
"reading a page tells no one else about it.",
|
|
src,
|
|
)
|
|
continue
|
|
if fetcher is None or url is None:
|
|
continue
|
|
if fetched >= MAX_IMAGES:
|
|
continue
|
|
fetched += 1
|
|
total += _check_image(report, target, src, fetcher)
|
|
|
|
if fetcher is not None and len(images) > MAX_IMAGES:
|
|
report.add(
|
|
"images-not-checked",
|
|
f"This page has {len(images)} images and the checker read the first "
|
|
f"{MAX_IMAGES}. Check the rest yourself.",
|
|
)
|
|
return total
|
|
|
|
|
|
def _check_image(report: Report, target: str, src: str, fetcher: Fetcher) -> int:
|
|
"""Fetch one image and check its format, size and metadata."""
|
|
try:
|
|
response = fetcher.get(target, cap=IMAGE_CAP)
|
|
except (UrlError, FetchError) as error:
|
|
report.add("image-unreachable", f"This image didn't load: {error}", src)
|
|
return 0
|
|
if response.status != OK:
|
|
report.add(
|
|
"image-unreachable",
|
|
f"This image didn't load (it returned {response.status}).",
|
|
src,
|
|
)
|
|
return 0
|
|
if "set-cookie" in response.headers:
|
|
report.add("cookie", "The server sets a cookie on this image.", src)
|
|
|
|
body = response.body
|
|
if response.truncated:
|
|
report.add(
|
|
"image-size",
|
|
f"This image is over {IMAGE_CAP // 1024} KB. Keep images under "
|
|
"50 KB, so a page loads quickly on a slow connection.",
|
|
src,
|
|
)
|
|
elif len(body) > IMAGE_SHOULD:
|
|
report.add(
|
|
"image-size",
|
|
f"This image is {len(body) // 1024} KB. Keep images under 50 KB, so "
|
|
"a page loads quickly on a slow connection.",
|
|
src,
|
|
)
|
|
|
|
if not any(body.startswith(magic) for magic in IMAGE_MAGIC):
|
|
report.add(
|
|
"image-format",
|
|
"This image isn't a GIF, JPEG or PNG. Use one of those, so every "
|
|
"client and old handset can show it.",
|
|
src,
|
|
)
|
|
elif _has_metadata(body):
|
|
report.add(
|
|
"image-metadata",
|
|
"This image carries camera or location metadata. Strip it before "
|
|
"publishing.",
|
|
src,
|
|
)
|
|
return len(body)
|
|
|
|
|
|
def _has_metadata(body: bytes) -> bool:
|
|
"""Say whether an image carries an Exif block."""
|
|
if body.startswith(b"\xff\xd8\xff"):
|
|
return b"Exif\x00\x00" in body[:4096]
|
|
if body.startswith(b"\x89PNG"):
|
|
return b"eXIf" in body[:4096]
|
|
return False
|
|
|
|
|
|
def _stylesheet_findings(report: Report, href: str, url: str, fetcher: Fetcher) -> None:
|
|
"""Fetch the linked stylesheet and compare it with the default one."""
|
|
target = urljoin(url, href)
|
|
host = urlsplit(url).hostname or ""
|
|
try:
|
|
response = fetcher.get(target, cap=CSS_CAP)
|
|
except (UrlError, FetchError) as error:
|
|
report.add(
|
|
"stylesheet-unreachable", f"The stylesheet didn't load: {error}", href
|
|
)
|
|
return
|
|
if response.status != OK:
|
|
report.add(
|
|
"stylesheet-unreachable",
|
|
f"The stylesheet didn't load (it returned {response.status}).",
|
|
href,
|
|
)
|
|
return
|
|
if "set-cookie" in response.headers:
|
|
report.add("cookie", "The server sets a cookie on the stylesheet.", href)
|
|
|
|
comparison = mews_css.compare(response.body.decode("utf-8", "replace"))
|
|
if comparison.has_import:
|
|
report.add(
|
|
"stylesheet-import",
|
|
"Your stylesheet copy uses @import. Remove it: a Mews page loads "
|
|
"one stylesheet and nothing else.",
|
|
href,
|
|
)
|
|
if not comparison.equal:
|
|
report.add(
|
|
"stylesheet-modified",
|
|
"Your copy of mews-0.1.css has been changed, starting around line "
|
|
f"{comparison.diff_line}. Copy it again unchanged. You may add "
|
|
"@font-face rules that load fonts from your own site.",
|
|
href,
|
|
)
|
|
for font in comparison.font_urls:
|
|
if not _is_same_site(host, urljoin(target, font)):
|
|
report.add(
|
|
"font-offsite",
|
|
"An @font-face rule loads a font from another site. Host the "
|
|
"font file on your own site.",
|
|
font,
|
|
)
|
|
|
|
|
|
# --- Entry points --------------------------------------------------------
|
|
|
|
|
|
def validate_bytes(
|
|
data: bytes,
|
|
*,
|
|
url: str | None = None,
|
|
response: Fetched | None = None,
|
|
fetcher: Fetcher | None = None,
|
|
) -> Report:
|
|
"""Validate one page.
|
|
|
|
url is the address the page was fetched from, which the same-site rules in
|
|
4.2 and 7.4 need. Without a fetcher the checks that need the live page are
|
|
skipped and said to be skipped.
|
|
"""
|
|
report = Report(url=url)
|
|
report.size = len(data)
|
|
|
|
can_parse, expand_entities = _prologue(report, data)
|
|
if not can_parse:
|
|
return report
|
|
|
|
try:
|
|
root = etree.fromstring(data, _parser(expand_entities=expand_entities))
|
|
except etree.XMLSyntaxError as error:
|
|
report.add(
|
|
"well-formed",
|
|
"This page isn't well-formed XML, so a Mews client can't read it "
|
|
f"({error.msg}).",
|
|
f"line {error.lineno}",
|
|
)
|
|
return report
|
|
tree = etree.ElementTree(root)
|
|
|
|
_document_findings(report, root)
|
|
_dtd_findings(report, tree)
|
|
|
|
host = urlsplit(url).hostname if url else None
|
|
head = root.find(f"{{{XHTML}}}head")
|
|
stylesheet, offsite = (None, False)
|
|
if head is not None:
|
|
stylesheet, offsite = _link_findings(report, head, host)
|
|
|
|
if response is not None:
|
|
_transport_findings(report, response)
|
|
|
|
image_bytes = _image_findings(report, root, url, fetcher)
|
|
|
|
if stylesheet and not offsite and url and fetcher is not None:
|
|
_stylesheet_findings(report, stylesheet, url, fetcher)
|
|
|
|
if (response is not None and response.truncated) or report.size > SIZE_MUST:
|
|
report.add(
|
|
"size-must",
|
|
f"The page markup is over {SIZE_MUST // 1024} KB. Split it into "
|
|
"several pages.",
|
|
)
|
|
elif report.size > SIZE_SHOULD:
|
|
report.add(
|
|
"size-should",
|
|
f"The page markup is {report.size // 1024} KB. Keep it under "
|
|
"64 KB, so it loads quickly on a slow connection.",
|
|
)
|
|
if fetcher is not None and report.size + image_bytes > SIZE_TOTAL:
|
|
report.add(
|
|
"size-total",
|
|
"The page and its images come to more than 320 KB together. Use "
|
|
"fewer or smaller images.",
|
|
)
|
|
|
|
if fetcher is None:
|
|
report.notes.append(
|
|
"The checks that need the live page were skipped: response headers "
|
|
"(7.1 to 7.4), image sizes and formats (4.2), and the stylesheet "
|
|
"copy (5.2)."
|
|
)
|
|
return report
|
|
|
|
|
|
def validate_file(path: str) -> Report:
|
|
"""Validate a page on disk."""
|
|
with open(path, "rb") as handle:
|
|
return validate_bytes(handle.read())
|
|
|
|
|
|
def validate_url(
|
|
url: str, *, fetcher: Fetcher | None = None, allow_loopback: bool = False
|
|
) -> Report:
|
|
"""Fetch a page and validate it, along with its stylesheet and images."""
|
|
owned = fetcher is None
|
|
client = fetcher or Fetcher(allow_loopback=allow_loopback)
|
|
try:
|
|
response = client.get(normalise_url(url))
|
|
if response.status != OK:
|
|
report = Report(url=response.url)
|
|
report.add(
|
|
"well-formed",
|
|
f"That page returned {response.status}. Check the address "
|
|
"and try again.",
|
|
)
|
|
return report
|
|
return validate_bytes(
|
|
response.body, url=response.url, response=response, fetcher=client
|
|
)
|
|
finally:
|
|
if owned:
|
|
client.close()
|
|
|
|
|
|
def main(argv: list[str] | None = None) -> int:
|
|
"""Check pages named on the command line, returning 1 if any doesn't conform."""
|
|
parser = argparse.ArgumentParser(
|
|
description="Check a page against Mews Profile 0.1 (see doc/SPEC.md)."
|
|
)
|
|
parser.add_argument("target", nargs="+", help="a page address or a file")
|
|
parser.add_argument(
|
|
"--local",
|
|
action="store_true",
|
|
help="also read addresses on this machine, for checking a site before "
|
|
"it is published",
|
|
)
|
|
parser.add_argument(
|
|
"--quiet", action="store_true", help="print nothing for pages that conform"
|
|
)
|
|
args = parser.parse_args(argv)
|
|
|
|
worst = 0
|
|
for target in args.target:
|
|
if "://" in target or not _looks_like_path(target):
|
|
try:
|
|
report = validate_url(target, allow_loopback=args.local)
|
|
except (UrlError, FetchError) as error:
|
|
print(f"{target} — {error}", file=sys.stderr)
|
|
worst = 1
|
|
continue
|
|
else:
|
|
report = validate_file(target)
|
|
|
|
if report.conforms and args.quiet:
|
|
continue
|
|
worst = max(worst, 0 if report.conforms else 1)
|
|
verdict = "conforms" if report.conforms else "doesn't conform"
|
|
print(f"{target} — {verdict}")
|
|
for finding in report.failures:
|
|
print(f" {finding}")
|
|
if report.warnings:
|
|
print(" Worth fixing:")
|
|
for finding in report.warnings:
|
|
print(f" {finding}")
|
|
for note in report.notes:
|
|
print(f" Note: {note}")
|
|
return worst
|
|
|
|
|
|
def _looks_like_path(target: str) -> bool:
|
|
"""Say whether a bare argument names a file rather than a site."""
|
|
return Path(target).exists()
|