"""The Mews page validator. SPEC.md 9.1 defines conformance as two checks: the page validates against dtd/mews-0.1.dtd, and it follows the rules a DTD cannot express. This module runs both and reports every failure with its spec section. Failures at MUST level mean the page does not conform. Failures at SHOULD level are warnings: they are worth fixing but do not make a page non-conforming. """ import argparse from dataclasses import dataclass, field import html.entities from importlib import resources from pathlib import Path import re import sys from urllib.parse import urljoin, urlsplit from lxml import etree from mews import VERSION, css as mews_css from mews.fetch import ( CSS_CAP, IMAGE_CAP, Fetched, Fetcher, FetchError, UrlError, normalise_url, same_site, ) XHTML = "http://www.w3.org/1999/xhtml" XML_LANG = "{http://www.w3.org/XML/1998/namespace}lang" MARKER = "mews-profile" CANONICAL_STYLESHEET = "https://mews.page/mews-0.1.css" OK = 200 MAX_FINDINGS = 50 MAX_IMAGES = 20 SIZE_MUST = 256 * 1024 SIZE_SHOULD = 64 * 1024 SIZE_TOTAL = 320 * 1024 IMAGE_SHOULD = 50 * 1024 XML_DECLARATION = re.compile(rb'^<\?xml version="1\.0" encoding="(?i:UTF-8)"\?>') DOCTYPE = re.compile( rb'' ) ALLOWED_META = ("mews-profile", "description", "author", "viewport") ALLOWED_RELS = ("stylesheet", "icon", "alternate") # Magic bytes for the three formats SPEC.md 4.2 recommends. IMAGE_MAGIC = { b"GIF87a": "GIF", b"GIF89a": "GIF", b"\x89PNG\r\n\x1a\n": "PNG", b"\xff\xd8\xff": "JPEG", } @dataclass(frozen=True) class Check: """One thing the validator looks at, and the spec section it comes from.""" code: str section: str level: str what: str CHECKS: tuple[Check, ...] = ( Check("xml-declaration", "3.1", "must", "UTF-8 XML declaration, no BOM"), Check("doctype", "3.1", "must", "the XHTML-MP 1.2 doctype, exactly"), Check("internal-subset", "3.1", "must", "no internal DTD subset"), Check("well-formed", "3", "must", "the page parses as XML"), Check("root-element", "3.1", "must", "html root in the XHTML namespace"), Check("lang", "3.1", "should", "xml:lang and lang agree on html"), Check("dtd-element", "4.1", "must", "only elements from 4.1"), Check("dtd-attribute", "4.1", "must", "only attributes from 4.1"), Check("dtd-nesting", "4.1", "must", "elements nest as the DTD allows"), Check("dtd-content", "4.1", "must", "element content follows the DTD"), Check("dtd-value", "4.1", "must", "attribute values from the permitted set"), Check("dtd-required-attribute", "4.1", "must", "required attributes present"), Check("dtd-head", "3.3", "must", "the head holds only what 3.3 permits"), Check("dtd-nested-table", "4.2", "must", "tables are not nested"), Check("dtd-variant", "5.3", "must", "body class is one of the variants"), Check("dtd-styling", "5", "must", "no author styling"), Check("dtd-other", "4.1", "must", "the page matches the Mews DTD"), Check("marker", "3.2", "must", "the conformance marker is present once"), Check("marker-version", "3.2", "must", "the marker names a known version"), Check("meta-name", "3.3", "must", "only the meta names 3.3 permits"), Check("link-rel", "3.3", "must", "only the link kinds 3.3 permits"), Check("link-stylesheet-count", "3.3", "must", "at most one stylesheet link"), Check("link-alternate-type", "3.3", "must", "feed links are Atom"), Check("icon-offsite", "3.3", "must", "the icon is on the page's own site"), Check("viewport", "3.3", "should", "the viewport meta is present"), Check("size-must", "4.3", "must", "markup under 256 KB"), Check("size-should", "4.3", "should", "markup under 64 KB"), Check("size-total", "4.3", "should", "page plus images under 320 KB"), Check("image-data-uri", "4.2", "must", "no data: URI images"), Check("image-offsite", "4.2", "must", "images are on the page's own site"), Check("image-format", "4.2", "should", "images are GIF, JPEG or PNG"), Check("image-size", "4.2", "should", "each image under 50 KB"), Check("image-unreachable", "4.2", "should", "images load"), Check("image-metadata", "4.2", "should", "images carry no camera metadata"), Check("images-not-checked", "4.2", "should", "how many images were checked"), Check("stylesheet-missing", "5.2", "should", "the default stylesheet is linked"), Check("stylesheet-canonical", "5.2", "should", "the site hosts its own copy"), Check("stylesheet-modified", "5.2", "must", "the copy is unmodified"), Check("stylesheet-import", "5.2", "must", "the copy has no @import"), Check("stylesheet-unreachable", "5.2", "should", "the stylesheet loads"), Check("font-offsite", "5.2", "must", "@font-face fonts are on the own site"), Check("stylesheet-offsite", "7.4", "must", "the stylesheet is on the own site"), Check("content-type", "7.1", "should", "a page content type"), Check("https", "7.2", "should", "served over HTTPS"), Check("validators", "7.3", "should", "Last-Modified or ETag is sent"), Check("cookie", "7.4", "must", "no cookies are set"), ) BY_CODE = {check.code: check for check in CHECKS} # Rules in sections 3 to 7 this validator deliberately leaves alone. Together # with CHECKS this covers every MUST and SHOULD in those sections; tests fail if # a section appears in neither. NOT_CHECKED = { "5.1": "The default stylesheet is compared byte for byte (5.2), which " "covers whether its base rules are valid WAP CSS.", "6.1": "Dated links are a client convention; the validator reads no feeds.", "6.2": "Atom feed contents are a client convention; only the link type is checked.", "7.2": "Whether a site redirects HTTP to HTTPS in a way old handsets can " "follow cannot be told from one request.", "4.2": "Whether text appears only inside an image cannot be told from markup.", } @dataclass(frozen=True) class Finding: """One rule a page broke.""" code: str message: str location: str = "" @property def section(self) -> str: """Return the spec section this finding comes from.""" return BY_CODE[self.code].section @property def level(self) -> str: """Return either must or should.""" return BY_CODE[self.code].level def __str__(self) -> str: where = f" ({self.location})" if self.location else "" return f"Section {self.section} — {self.message}{where}" @dataclass class Report: """What the validator found, plus the facts the directory needs.""" url: str | None = None findings: list[Finding] = field(default_factory=list) notes: list[str] = field(default_factory=list) title: str = "" description: str = "" language: str = "" size: int = 0 def add(self, code: str, message: str, location: str = "") -> None: """Record one finding, up to the report cap.""" if len(self.findings) < MAX_FINDINGS: self.findings.append(Finding(code, message, location)) @property def failures(self) -> list[Finding]: """Return the findings that make the page non-conforming.""" return [f for f in self.findings if f.level == "must"] @property def warnings(self) -> list[Finding]: """Return the findings worth fixing that still leave the page conforming.""" return [f for f in self.findings if f.level == "should"] @property def conforms(self) -> bool: """Say whether the page follows every MUST rule the validator checks.""" return not self.failures def _entity_declarations() -> str: """Declare the entity set the XHTML-MP doctype would have defined. SPEC.md 9.1 says clients must not fetch the doctype, and the local copy of the driver DTD pulls its modules from w3.org, so the entities are supplied from the stdlib's HTML 4 table instead — the same 252 names. """ return "".join( f'' for code, name in sorted(html.entities.codepoint2name.items()) ) class _Resolver(etree.Resolver): """Resolves the XHTML-MP doctype to entity declarations and nothing else.""" def __init__(self) -> None: self._entities = _entity_declarations() def resolve(self, system_url, public_id, context): """Answer the parser's request for an external entity.""" if public_id == "-//WAPFORUM//DTD XHTML Mobile 1.2//EN": return self.resolve_string(self._entities, context) # With no_network set, returning None makes any other doctype fail. return None def _parser(*, expand_entities: bool) -> etree.XMLParser: """Build an XML parser that never reaches the network.""" parser = etree.XMLParser( load_dtd=expand_entities, resolve_entities=expand_entities, no_network=True, dtd_validation=False, attribute_defaults=False, huge_tree=False, ) if expand_entities: parser.resolvers.add(_Resolver()) return parser _DTD: etree.DTD | None = None def mews_dtd() -> etree.DTD: """Return the Mews DTD, loaded once per process.""" global _DTD # noqa: PLW0603 if _DTD is None: with resources.as_file( resources.files("mews.data").joinpath("mews-0.1.dtd") ) as path: _DTD = etree.DTD(str(path)) return _DTD def local(element: etree._Element) -> str: """Return an element's name without its namespace.""" return etree.QName(element).localname # --- The DTD check ------------------------------------------------------- UNDECLARED_ELEMENT = re.compile(r"^No declaration for element (\S+)") UNDECLARED_ATTRIBUTE = re.compile( r"^No declaration for attribute (\S+) of element (\S+)" ) NOT_ALLOWED_IN = re.compile( r"^Element (\S+) is not declared in (\S+) list of possible children" ) CONTENT_MODEL = re.compile( r"^Element (\S+) content does not follow the DTD, expecting (.*?), got \(?(.*?)\)?$" ) BAD_VALUE = re.compile( r'^Value "(.*?)" for attribute (\S+) of (\S+) is not among the enumerated set' ) MISSING_ATTRIBUTE = re.compile(r"^Element (\S+) does not carry attribute (\S+)") def _enumerated_values(element: str, attribute: str) -> list[str]: """Return the values the DTD permits for an attribute, read from the DTD.""" for declared in mews_dtd().iterelements(): if declared.name != element: continue for attr in declared.iterattributes(): if attr.name == attribute: return list(attr.itervalues() or []) return [] def _dtd_findings(report: Report, tree: etree._ElementTree) -> None: """Validate against the Mews DTD and turn libxml2's errors into findings.""" dtd = mews_dtd() if dtd.validate(tree): return errors = [(entry.line, entry.message) for entry in dtd.error_log] undeclared = { match.group(1) for _, message in errors for match in [UNDECLARED_ELEMENT.match(message)] if match } for line, message in errors: finding = _map_error(message, undeclared) if finding is not None: code, text = finding report.add(code, text, f"line {line}") def _map_error(message: str, undeclared: set[str]) -> tuple[str, str] | None: """Map one libxml2 message to a code and a plain sentence, or drop it. One mistake makes libxml2 say several things: an unknown element is also an unknown attribute and a content-model break. Only the clearest is kept. """ match = UNDECLARED_ELEMENT.match(message) if match: name = match.group(1) if name == "style": return ( "dtd-styling", "The