mews.page/mews/lint.py

977 lines
34 KiB
Python
Raw Normal View History

"""The Mews page validator.
SPEC.md 9.1 defines conformance as two checks: the page validates against
dtd/mews-0.1.dtd, and it follows the rules a DTD cannot express. This module
runs both and reports every failure with its spec section.
Failures at MUST level mean the page does not conform. Failures at SHOULD level
are warnings: they are worth fixing but do not make a page non-conforming.
"""
import argparse
from dataclasses import dataclass, field
import html.entities
from importlib import resources
from pathlib import Path
import re
import sys
from urllib.parse import urljoin, urlsplit
from lxml import etree
from mews import VERSION, css as mews_css
from mews.fetch import (
CSS_CAP,
IMAGE_CAP,
Fetched,
Fetcher,
FetchError,
UrlError,
normalise_url,
same_site,
)
XHTML = "http://www.w3.org/1999/xhtml"
XML_LANG = "{http://www.w3.org/XML/1998/namespace}lang"
MARKER = "mews-profile"
CANONICAL_STYLESHEET = "https://mews.page/mews-0.1.css"
OK = 200
MAX_FINDINGS = 50
MAX_IMAGES = 20
SIZE_MUST = 256 * 1024
SIZE_SHOULD = 64 * 1024
SIZE_TOTAL = 320 * 1024
IMAGE_SHOULD = 50 * 1024
XML_DECLARATION = re.compile(rb'^<\?xml version="1\.0" encoding="(?i:UTF-8)"\?>')
DOCTYPE = re.compile(
rb'<!DOCTYPE\s+html\s+PUBLIC\s+"-//WAPFORUM//DTD XHTML Mobile 1\.2//EN"\s+'
rb'"http://www\.openmobilealliance\.org/tech/DTD/xhtml-mobile12\.dtd"\s*>'
)
ALLOWED_META = ("mews-profile", "description", "author", "viewport")
ALLOWED_RELS = ("stylesheet", "icon", "alternate")
# Magic bytes for the three formats SPEC.md 4.2 recommends.
IMAGE_MAGIC = {
b"GIF87a": "GIF",
b"GIF89a": "GIF",
b"\x89PNG\r\n\x1a\n": "PNG",
b"\xff\xd8\xff": "JPEG",
}
@dataclass(frozen=True)
class Check:
"""One thing the validator looks at, and the spec section it comes from."""
code: str
section: str
level: str
what: str
CHECKS: tuple[Check, ...] = (
Check("xml-declaration", "3.1", "must", "UTF-8 XML declaration, no BOM"),
Check("doctype", "3.1", "must", "the XHTML-MP 1.2 doctype, exactly"),
Check("internal-subset", "3.1", "must", "no internal DTD subset"),
Check("well-formed", "3", "must", "the page parses as XML"),
Check("root-element", "3.1", "must", "html root in the XHTML namespace"),
Check("lang", "3.1", "should", "xml:lang and lang agree on html"),
Check("dtd-element", "4.1", "must", "only elements from 4.1"),
Check("dtd-attribute", "4.1", "must", "only attributes from 4.1"),
Check("dtd-nesting", "4.1", "must", "elements nest as the DTD allows"),
Check("dtd-content", "4.1", "must", "element content follows the DTD"),
Check("dtd-value", "4.1", "must", "attribute values from the permitted set"),
Check("dtd-required-attribute", "4.1", "must", "required attributes present"),
Check("dtd-head", "3.3", "must", "the head holds only what 3.3 permits"),
Check("dtd-nested-table", "4.2", "must", "tables are not nested"),
Check("dtd-variant", "5.3", "must", "body class is one of the variants"),
Check("dtd-styling", "5", "must", "no author styling"),
Check("dtd-other", "4.1", "must", "the page matches the Mews DTD"),
Check("marker", "3.2", "must", "the conformance marker is present once"),
Check("marker-version", "3.2", "must", "the marker names a known version"),
Check("meta-name", "3.3", "must", "only the meta names 3.3 permits"),
Check("link-rel", "3.3", "must", "only the link kinds 3.3 permits"),
Check("link-stylesheet-count", "3.3", "must", "at most one stylesheet link"),
Check("link-alternate-type", "3.3", "must", "feed links are Atom"),
Check("icon-offsite", "3.3", "must", "the icon is on the page's own site"),
Check("viewport", "3.3", "should", "the viewport meta is present"),
Check("size-must", "4.3", "must", "markup under 256 KB"),
Check("size-should", "4.3", "should", "markup under 64 KB"),
Check("size-total", "4.3", "should", "page plus images under 320 KB"),
Check("image-data-uri", "4.2", "must", "no data: URI images"),
Check("image-offsite", "4.2", "must", "images are on the page's own site"),
Check("image-format", "4.2", "should", "images are GIF, JPEG or PNG"),
Check("image-size", "4.2", "should", "each image under 50 KB"),
Check("image-unreachable", "4.2", "should", "images load"),
Check("image-metadata", "4.2", "should", "images carry no camera metadata"),
Check("images-not-checked", "4.2", "should", "how many images were checked"),
Check("stylesheet-missing", "5.2", "should", "the default stylesheet is linked"),
Check("stylesheet-canonical", "5.2", "should", "the site hosts its own copy"),
Check("stylesheet-modified", "5.2", "must", "the copy is unmodified"),
Check("stylesheet-import", "5.2", "must", "the copy has no @import"),
Check("stylesheet-unreachable", "5.2", "should", "the stylesheet loads"),
Check("font-offsite", "5.2", "must", "@font-face fonts are on the own site"),
Check("stylesheet-offsite", "7.4", "must", "the stylesheet is on the own site"),
Check("content-type", "7.1", "should", "a page content type"),
Check("https", "7.2", "should", "served over HTTPS"),
Check("validators", "7.3", "should", "Last-Modified or ETag is sent"),
Check("cookie", "7.4", "must", "no cookies are set"),
)
BY_CODE = {check.code: check for check in CHECKS}
# Rules in sections 3 to 7 this validator deliberately leaves alone. Together
# with CHECKS this covers every MUST and SHOULD in those sections; tests fail if
# a section appears in neither.
NOT_CHECKED = {
"5.1": "The default stylesheet is compared byte for byte (5.2), which "
"covers whether its base rules are valid WAP CSS.",
"6.1": "Dated links are a client convention; the validator reads no feeds.",
"6.2": "Atom feed contents are a client convention; only the link type is checked.",
"7.2": "Whether a site redirects HTTP to HTTPS in a way old handsets can "
"follow cannot be told from one request.",
"4.2": "Whether text appears only inside an image cannot be told from markup.",
}
@dataclass(frozen=True)
class Finding:
"""One rule a page broke."""
code: str
message: str
location: str = ""
@property
def section(self) -> str:
"""Return the spec section this finding comes from."""
return BY_CODE[self.code].section
@property
def level(self) -> str:
"""Return either must or should."""
return BY_CODE[self.code].level
def __str__(self) -> str:
where = f" ({self.location})" if self.location else ""
return f"Section {self.section} — {self.message}{where}"
@dataclass
class Report:
"""What the validator found, plus the facts the directory needs."""
url: str | None = None
findings: list[Finding] = field(default_factory=list)
notes: list[str] = field(default_factory=list)
title: str = ""
description: str = ""
language: str = ""
size: int = 0
def add(self, code: str, message: str, location: str = "") -> None:
"""Record one finding, up to the report cap."""
if len(self.findings) < MAX_FINDINGS:
self.findings.append(Finding(code, message, location))
@property
def failures(self) -> list[Finding]:
"""Return the findings that make the page non-conforming."""
return [f for f in self.findings if f.level == "must"]
@property
def warnings(self) -> list[Finding]:
"""Return the findings worth fixing that still leave the page conforming."""
return [f for f in self.findings if f.level == "should"]
@property
def conforms(self) -> bool:
"""Say whether the page follows every MUST rule the validator checks."""
return not self.failures
def _entity_declarations() -> str:
"""Declare the entity set the XHTML-MP doctype would have defined.
SPEC.md 9.1 says clients must not fetch the doctype, and the local copy of
the driver DTD pulls its modules from w3.org, so the entities are supplied
from the stdlib's HTML 4 table instead — the same 252 names.
"""
return "".join(
f'<!ENTITY {name} "&#{code};">'
for code, name in sorted(html.entities.codepoint2name.items())
)
class _Resolver(etree.Resolver):
"""Resolves the XHTML-MP doctype to entity declarations and nothing else."""
def __init__(self) -> None:
self._entities = _entity_declarations()
def resolve(self, system_url, public_id, context):
"""Answer the parser's request for an external entity."""
if public_id == "-//WAPFORUM//DTD XHTML Mobile 1.2//EN":
return self.resolve_string(self._entities, context)
# With no_network set, returning None makes any other doctype fail.
return None
def _parser(*, expand_entities: bool) -> etree.XMLParser:
"""Build an XML parser that never reaches the network."""
parser = etree.XMLParser(
load_dtd=expand_entities,
resolve_entities=expand_entities,
no_network=True,
dtd_validation=False,
attribute_defaults=False,
huge_tree=False,
)
if expand_entities:
parser.resolvers.add(_Resolver())
return parser
_DTD: etree.DTD | None = None
def mews_dtd() -> etree.DTD:
"""Return the Mews DTD, loaded once per process."""
global _DTD # noqa: PLW0603
if _DTD is None:
with resources.as_file(
resources.files("mews.data").joinpath("mews-0.1.dtd")
) as path:
_DTD = etree.DTD(str(path))
return _DTD
def local(element: etree._Element) -> str:
"""Return an element's name without its namespace."""
return etree.QName(element).localname
# --- The DTD check -------------------------------------------------------
UNDECLARED_ELEMENT = re.compile(r"^No declaration for element (\S+)")
UNDECLARED_ATTRIBUTE = re.compile(
r"^No declaration for attribute (\S+) of element (\S+)"
)
NOT_ALLOWED_IN = re.compile(
r"^Element (\S+) is not declared in (\S+) list of possible children"
)
CONTENT_MODEL = re.compile(
r"^Element (\S+) content does not follow the DTD, expecting (.*?), got \(?(.*?)\)?$"
)
BAD_VALUE = re.compile(
r'^Value "(.*?)" for attribute (\S+) of (\S+) is not among the enumerated set'
)
MISSING_ATTRIBUTE = re.compile(r"^Element (\S+) does not carry attribute (\S+)")
def _enumerated_values(element: str, attribute: str) -> list[str]:
"""Return the values the DTD permits for an attribute, read from the DTD."""
for declared in mews_dtd().iterelements():
if declared.name != element:
continue
for attr in declared.iterattributes():
if attr.name == attribute:
return list(attr.itervalues() or [])
return []
def _dtd_findings(report: Report, tree: etree._ElementTree) -> None:
"""Validate against the Mews DTD and turn libxml2's errors into findings."""
dtd = mews_dtd()
if dtd.validate(tree):
return
errors = [(entry.line, entry.message) for entry in dtd.error_log]
undeclared = {
match.group(1)
for _, message in errors
for match in [UNDECLARED_ELEMENT.match(message)]
if match
}
for line, message in errors:
finding = _map_error(message, undeclared)
if finding is not None:
code, text = finding
report.add(code, text, f"line {line}")
def _map_error(message: str, undeclared: set[str]) -> tuple[str, str] | None:
"""Map one libxml2 message to a code and a plain sentence, or drop it.
One mistake makes libxml2 say several things: an unknown element is also an
unknown attribute and a content-model break. Only the clearest is kept.
"""
match = UNDECLARED_ELEMENT.match(message)
if match:
name = match.group(1)
if name == "style":
return (
"dtd-styling",
"The <style> element isn't allowed. Mews pages carry no author "
"styling, so readers control how a page looks.",
)
if name == "script":
return (
"dtd-element",
"The <script> element isn't allowed. Mews clients run no "
"scripts, so a page has to work without them.",
)
return (
"dtd-element",
f"The <{name}> element isn't allowed. Remove it, or use an element "
"from section 4.1.",
)
match = UNDECLARED_ATTRIBUTE.match(message)
if match:
attribute, element = match.group(1), match.group(2)
if element in undeclared:
return None
if attribute == "style":
return (
"dtd-styling",
f"The style attribute isn't allowed on <{element}>. Mews pages "
"carry no author styling.",
)
return (
"dtd-attribute",
f"The {attribute} attribute isn't allowed on <{element}>. Remove it.",
)
match = NOT_ALLOWED_IN.match(message)
if match:
child, parent = match.group(1), match.group(2)
if child in undeclared:
return None
if child == "table":
return (
"dtd-nested-table",
"Tables can't be nested. Move the inner table out, or use a list.",
)
return ("dtd-nesting", f"<{child}> can't go inside <{parent}>.")
match = CONTENT_MODEL.match(message)
if match:
element, got = match.group(1), match.group(3).split()
if any(name in undeclared for name in got):
return None
if element == "head":
return (
"dtd-head",
"The head can hold only a title, then meta and link elements, "
"with the title first.",
)
if not got:
if element == "body":
return ("dtd-content", "The page has nothing in its body.")
return (
"dtd-content",
f"<{element}> is empty. Either fill it or remove it.",
)
return (
"dtd-content",
f"<{element}> can't hold what's inside it here.",
)
match = BAD_VALUE.match(message)
if match:
value, attribute, element = match.groups()
if (element, attribute) == ("link", "rel"):
return (
"link-rel",
f'The link with rel="{value}" isn\'t allowed. A head may hold '
"the stylesheet link, an icon and a feed link.",
)
if (element, attribute) == ("body", "class"):
return (
"dtd-variant",
f'"{value}" isn\'t a body variant. Use mews-warm, mews-cool, '
"mews-green or mews-mono, or leave class out.",
)
allowed = ", ".join(_enumerated_values(element, attribute))
return (
"dtd-value",
f'"{value}" isn\'t allowed for {attribute} on <{element}>. '
+ (f"Use one of: {allowed}." if allowed else "Remove it."),
)
match = MISSING_ATTRIBUTE.match(message)
if match:
element, attribute = match.groups()
if element == "img" and attribute == "alt":
return (
"dtd-required-attribute",
'Every image needs an alt attribute: a description, or alt="" '
"when the image is decoration.",
)
return (
"dtd-required-attribute",
f"<{element}> needs a {attribute} attribute.",
)
return ("dtd-other", f"This page doesn't match the Mews DTD ({message}).")
# --- The rule checks -----------------------------------------------------
def _prologue(report: Report, data: bytes) -> tuple[bool, bool]:
"""Check the bytes before the root element.
Returns whether the page can be parsed at all and whether its entities may
be expanded. This runs first on purpose: a page carrying its own entity
declarations is refused before any parser sees them.
"""
head = data[:2048]
if head.startswith(b"\xef\xbb\xbf"):
report.add(
"xml-declaration",
"Remove the byte order mark at the start of the file. A Mews page "
"starts with the XML declaration.",
)
head = head[3:]
if not XML_DECLARATION.match(head):
report.add(
"xml-declaration",
'Start the page with <?xml version="1.0" encoding="UTF-8"?>.',
)
start = head.find(b"<!DOCTYPE")
if start == -1:
report.add(
"doctype",
"Add the XHTML Mobile 1.2 doctype after the XML declaration, as "
"section 3.1 shows.",
)
return True, False
end = head.find(b">", start)
if b"[" in head[start : end if end != -1 else len(head)]:
report.add(
"internal-subset",
"Remove the extra declarations from the doctype. A Mews page uses "
"the doctype from section 3.1 and nothing else.",
)
return False, False
if not DOCTYPE.match(head, start):
report.add(
"doctype",
"Use the doctype from section 3.1 exactly, pointing at the XHTML "
"Mobile 1.2 DTD.",
)
return True, False
return True, True
def _document_findings(report: Report, root: etree._Element) -> None:
"""Check the root element and the head against sections 3.1 to 3.3."""
if root.tag != f"{{{XHTML}}}html":
report.add(
"root-element",
'The root element must be <html xmlns="http://www.w3.org/1999/xhtml">.',
)
return
xml_lang, lang = root.get(XML_LANG), root.get("lang")
if not xml_lang or not lang or xml_lang != lang:
report.add(
"lang",
"Name the page language with matching xml:lang and lang attributes "
"on <html>, so browsers and clients both know it.",
)
report.language = xml_lang or lang or ""
head = root.find(f"{{{XHTML}}}head")
if head is None:
return
title = head.find(f"{{{XHTML}}}title")
if title is not None and title.text:
report.title = _clean(title.text)
markers = []
for meta in head.iterfind(f"{{{XHTML}}}meta"):
name = meta.get("name") or ""
if name not in ALLOWED_META:
report.add(
"meta-name",
f'The meta element named "{name}" isn\'t allowed. Section 3.3 '
"lists the ones a head may hold.",
)
continue
if name == MARKER:
markers.append(meta.get("content") or "")
elif name == "description":
report.description = _clean(meta.get("content") or "")
if len(markers) > 1:
report.add(
"marker",
'The head carries the <meta name="mews-profile" /> marker '
f"{len(markers)} times. Keep one.",
)
elif not markers:
report.add(
"marker",
'Add <meta name="mews-profile" content="0.1" /> to the head. '
"Clients use it to tell a Mews page apart.",
)
elif markers[0] != VERSION:
report.add(
"marker-version",
f'The marker says version "{markers[0]}". This validator checks '
f"version {VERSION}.",
)
viewport = any(
meta.get("name") == "viewport" for meta in head.iterfind(f"{{{XHTML}}}meta")
)
if not viewport:
report.add(
"viewport",
'Add <meta name="viewport" content="width=device-width" /> so phones '
"show the page at a readable size.",
)
def _link_findings(
report: Report, head: etree._Element, host: str | None
) -> tuple[str | None, bool]:
"""Check the head's links, returning the stylesheet href and if it's off-site."""
stylesheets: list[str] = []
for link in head.iterfind(f"{{{XHTML}}}link"):
rel = (link.get("rel") or "").lower()
href = link.get("href") or ""
if rel not in ALLOWED_RELS:
# Reported by the DTD check, which also names the three kinds.
continue
if rel == "stylesheet":
stylesheets.append(href)
elif rel == "alternate":
kind = (link.get("type") or "").split(";")[0].strip()
if kind != "application/atom+xml":
report.add(
"link-alternate-type",
'A feed link needs type="application/atom+xml".',
)
elif rel == "icon" and host and not _is_same_site(host, href):
report.add(
"icon-offsite",
"The icon is on another site. Host it on your own site, so "
"reading a page tells no one else about it.",
href,
)
if len(stylesheets) > 1:
report.add(
"link-stylesheet-count",
"Link the default stylesheet once. A Mews page has no other stylesheet.",
)
if not stylesheets:
report.add(
"stylesheet-missing",
"Link your copy of mews-0.1.css, so readers without a Mews client "
"still get good presentation.",
)
return None, False
href = stylesheets[0]
if host and not _is_same_site(host, href):
if _absolute(href).rstrip("/") == CANONICAL_STYLESHEET:
report.add(
"stylesheet-canonical",
"Copy mews-0.1.css to your own site and link that copy, so no "
"single server sees traffic across every Mews site.",
href,
)
else:
report.add(
"stylesheet-offsite",
"The stylesheet is on another site. A Mews page loads nothing "
"from other sites. Copy mews-0.1.css to your own site.",
href,
)
return href, True
return href, False
def _clean(text: str) -> str:
"""Collapse whitespace and drop characters that could reshape a listing."""
# Bidirectional overrides are printable but can make a title read as a
# different domain, so they go too.
overrides = frozenset(
chr(code) for code in (*range(0x202A, 0x202F), *range(0x2066, 0x206A))
)
stripped = "".join(
character
for character in text
if character.isprintable() and character not in overrides
)
return " ".join(stripped.split())
def _is_same_site(host: str, href: str) -> bool:
"""Say whether an href stays on the page's registered domain."""
other = urlsplit(href).hostname
return other is None or same_site(host, other)
def _absolute(href: str) -> str:
"""Treat an href with no scheme as https, for comparing with a known URL."""
return href if "://" in href else "https://" + href.lstrip("/")
def _transport_findings(report: Report, response: Fetched) -> None:
"""Check the response headers against sections 7.1 to 7.4."""
if "set-cookie" in response.headers:
report.add(
"cookie",
"The server sets a cookie on this page. A Mews page sets cookies "
"only for a form the reader submits.",
)
kind = (response.headers.get("content-type") or "").split(";")[0].strip().lower()
if kind not in (
"text/html",
"application/xhtml+xml",
"application/vnd.wap.xhtml+xml",
):
report.add(
"content-type",
f'The server sends this page as "{kind or "nothing"}". Send it as '
"text/html, so a browser still shows a page with a small error.",
)
if not response.headers.get("last-modified") and not response.headers.get("etag"):
report.add(
"validators",
"Send a Last-Modified or ETag header, so clients and the directory "
"can check for changes cheaply.",
)
if urlsplit(response.url).scheme != "https" or response.scheme_downgraded:
report.add(
"https",
"Serve the page over HTTPS. You may serve plain HTTP in parallel "
"for old handsets.",
)
def _image_findings(
report: Report, root: etree._Element, url: str | None, fetcher: Fetcher | None
) -> int:
"""Check every image, returning the bytes the fetched ones took."""
host = urlsplit(url).hostname if url else None
images = list(root.iter(f"{{{XHTML}}}img"))
total = 0
fetched = 0
for image in images:
src = (image.get("src") or "").strip()
if src.lower().startswith("data:"):
report.add(
"image-data-uri",
"This image is built into the page as a data: URI. Save it as a "
"file on your own site and link it.",
)
continue
if not src:
continue
target = urljoin(url, src) if url else src
if host and not _is_same_site(host, target):
report.add(
"image-offsite",
"This image is on another site. Host it on your own site, so "
"reading a page tells no one else about it.",
src,
)
continue
if fetcher is None or url is None:
continue
if fetched >= MAX_IMAGES:
continue
fetched += 1
total += _check_image(report, target, src, fetcher)
if fetcher is not None and len(images) > MAX_IMAGES:
report.add(
"images-not-checked",
f"This page has {len(images)} images and the checker read the first "
f"{MAX_IMAGES}. Check the rest yourself.",
)
return total
def _check_image(report: Report, target: str, src: str, fetcher: Fetcher) -> int:
"""Fetch one image and check its format, size and metadata."""
try:
response = fetcher.get(target, cap=IMAGE_CAP)
except (UrlError, FetchError) as error:
report.add("image-unreachable", f"This image didn't load: {error}", src)
return 0
if response.status != OK:
report.add(
"image-unreachable",
f"This image didn't load (it returned {response.status}).",
src,
)
return 0
if "set-cookie" in response.headers:
report.add("cookie", "The server sets a cookie on this image.", src)
body = response.body
if response.truncated:
report.add(
"image-size",
f"This image is over {IMAGE_CAP // 1024} KB. Keep images under "
"50 KB, so a page loads quickly on a slow connection.",
src,
)
elif len(body) > IMAGE_SHOULD:
report.add(
"image-size",
f"This image is {len(body) // 1024} KB. Keep images under 50 KB, so "
"a page loads quickly on a slow connection.",
src,
)
if not any(body.startswith(magic) for magic in IMAGE_MAGIC):
report.add(
"image-format",
"This image isn't a GIF, JPEG or PNG. Use one of those, so every "
"client and old handset can show it.",
src,
)
elif _has_metadata(body):
report.add(
"image-metadata",
"This image carries camera or location metadata. Strip it before "
"publishing.",
src,
)
return len(body)
def _has_metadata(body: bytes) -> bool:
"""Say whether an image carries an Exif block."""
if body.startswith(b"\xff\xd8\xff"):
return b"Exif\x00\x00" in body[:4096]
if body.startswith(b"\x89PNG"):
return b"eXIf" in body[:4096]
return False
def _stylesheet_findings(report: Report, href: str, url: str, fetcher: Fetcher) -> None:
"""Fetch the linked stylesheet and compare it with the default one."""
target = urljoin(url, href)
host = urlsplit(url).hostname or ""
try:
response = fetcher.get(target, cap=CSS_CAP)
except (UrlError, FetchError) as error:
report.add(
"stylesheet-unreachable", f"The stylesheet didn't load: {error}", href
)
return
if response.status != OK:
report.add(
"stylesheet-unreachable",
f"The stylesheet didn't load (it returned {response.status}).",
href,
)
return
if "set-cookie" in response.headers:
report.add("cookie", "The server sets a cookie on the stylesheet.", href)
comparison = mews_css.compare(response.body.decode("utf-8", "replace"))
if comparison.has_import:
report.add(
"stylesheet-import",
"Your stylesheet copy uses @import. Remove it: a Mews page loads "
"one stylesheet and nothing else.",
href,
)
if not comparison.equal:
report.add(
"stylesheet-modified",
"Your copy of mews-0.1.css has been changed, starting around line "
f"{comparison.diff_line}. Copy it again unchanged. You may add "
"@font-face rules that load fonts from your own site.",
href,
)
for font in comparison.font_urls:
if not _is_same_site(host, urljoin(target, font)):
report.add(
"font-offsite",
"An @font-face rule loads a font from another site. Host the "
"font file on your own site.",
font,
)
# --- Entry points --------------------------------------------------------
def validate_bytes(
data: bytes,
*,
url: str | None = None,
response: Fetched | None = None,
fetcher: Fetcher | None = None,
) -> Report:
"""Validate one page.
url is the address the page was fetched from, which the same-site rules in
4.2 and 7.4 need. Without a fetcher the checks that need the live page are
skipped and said to be skipped.
"""
report = Report(url=url)
report.size = len(data)
can_parse, expand_entities = _prologue(report, data)
if not can_parse:
return report
try:
root = etree.fromstring(data, _parser(expand_entities=expand_entities))
except etree.XMLSyntaxError as error:
report.add(
"well-formed",
"This page isn't well-formed XML, so a Mews client can't read it "
f"({error.msg}).",
f"line {error.lineno}",
)
return report
tree = etree.ElementTree(root)
_document_findings(report, root)
_dtd_findings(report, tree)
host = urlsplit(url).hostname if url else None
head = root.find(f"{{{XHTML}}}head")
stylesheet, offsite = (None, False)
if head is not None:
stylesheet, offsite = _link_findings(report, head, host)
if response is not None:
_transport_findings(report, response)
image_bytes = _image_findings(report, root, url, fetcher)
if stylesheet and not offsite and url and fetcher is not None:
_stylesheet_findings(report, stylesheet, url, fetcher)
if (response is not None and response.truncated) or report.size > SIZE_MUST:
report.add(
"size-must",
f"The page markup is over {SIZE_MUST // 1024} KB. Split it into "
"several pages.",
)
elif report.size > SIZE_SHOULD:
report.add(
"size-should",
f"The page markup is {report.size // 1024} KB. Keep it under "
"64 KB, so it loads quickly on a slow connection.",
)
if fetcher is not None and report.size + image_bytes > SIZE_TOTAL:
report.add(
"size-total",
"The page and its images come to more than 320 KB together. Use "
"fewer or smaller images.",
)
if fetcher is None:
report.notes.append(
"The checks that need the live page were skipped: response headers "
"(7.1 to 7.4), image sizes and formats (4.2), and the stylesheet "
"copy (5.2)."
)
return report
def validate_file(path: str) -> Report:
"""Validate a page on disk."""
with open(path, "rb") as handle:
return validate_bytes(handle.read())
def validate_url(
url: str, *, fetcher: Fetcher | None = None, allow_loopback: bool = False
) -> Report:
"""Fetch a page and validate it, along with its stylesheet and images."""
owned = fetcher is None
client = fetcher or Fetcher(allow_loopback=allow_loopback)
try:
response = client.get(normalise_url(url))
if response.status != OK:
report = Report(url=response.url)
report.add(
"well-formed",
f"That page returned {response.status}. Check the address "
"and try again.",
)
return report
return validate_bytes(
response.body, url=response.url, response=response, fetcher=client
)
finally:
if owned:
client.close()
def main(argv: list[str] | None = None) -> int:
"""Check pages named on the command line, returning 1 if any doesn't conform."""
parser = argparse.ArgumentParser(
description="Check a page against Mews Profile 0.1 (see doc/SPEC.md)."
)
parser.add_argument("target", nargs="+", help="a page address or a file")
parser.add_argument(
"--local",
action="store_true",
help="also read addresses on this machine, for checking a site before "
"it is published",
)
parser.add_argument(
"--quiet", action="store_true", help="print nothing for pages that conform"
)
args = parser.parse_args(argv)
worst = 0
for target in args.target:
if "://" in target or not _looks_like_path(target):
try:
report = validate_url(target, allow_loopback=args.local)
except (UrlError, FetchError) as error:
print(f"{target} — {error}", file=sys.stderr)
worst = 1
continue
else:
report = validate_file(target)
if report.conforms and args.quiet:
continue
worst = max(worst, 0 if report.conforms else 1)
verdict = "conforms" if report.conforms else "doesn't conform"
print(f"{target} — {verdict}")
for finding in report.failures:
print(f" {finding}")
if report.warnings:
print(" Worth fixing:")
for finding in report.warnings:
print(f" {finding}")
for note in report.notes:
print(f" Note: {note}")
return worst
def _looks_like_path(target: str) -> bool:
"""Say whether a bare argument names a file rather than a site."""
return Path(target).exists()