feat: page validator for both section 9.1 checks
This commit is contained in:
parent
99dc9f7a88
commit
b3c436bf1a
13 changed files with 2703 additions and 0 deletions
6
mews/__init__.py
Normal file
6
mews/__init__.py
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
"""Page validator and site directory for the Mews Profile.
|
||||
|
||||
See doc/SPEC.md for the format this package checks.
|
||||
"""
|
||||
|
||||
VERSION = "0.1"
|
||||
136
mews/css.py
Normal file
136
mews/css.py
Normal file
|
|
@ -0,0 +1,136 @@
|
|||
"""Comparing a site's stylesheet against the default one.
|
||||
|
||||
SPEC.md 5.2 lets a site host its own copy of the default stylesheet and change
|
||||
nothing in it, except to add @font-face rules that load fonts from the site
|
||||
itself. Checking that means comparing text, since a site may legitimately
|
||||
differ in line endings and trailing whitespace after a copy-paste.
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
import difflib
|
||||
from functools import cache
|
||||
from importlib import resources
|
||||
import re
|
||||
|
||||
FONT_FACE = re.compile(r"@font-face\b", re.IGNORECASE)
|
||||
URL_IN_SRC = re.compile(r"url\(\s*['\"]?([^'\")]+)", re.IGNORECASE)
|
||||
IMPORT_RULE = re.compile(r"@import\b", re.IGNORECASE)
|
||||
|
||||
|
||||
@dataclass
|
||||
class Comparison:
|
||||
"""What a candidate stylesheet differs from the default by."""
|
||||
|
||||
equal: bool
|
||||
diff_line: int | None = None
|
||||
diff: str = ""
|
||||
font_faces: list[str] = field(default_factory=list)
|
||||
has_import: bool = False
|
||||
|
||||
@property
|
||||
def font_urls(self) -> list[str]:
|
||||
"""Every url() the added @font-face rules load."""
|
||||
return [
|
||||
match.group(1).strip()
|
||||
for block in self.font_faces
|
||||
for match in URL_IN_SRC.finditer(block)
|
||||
]
|
||||
|
||||
|
||||
@cache
|
||||
def canonical() -> str:
|
||||
"""Return the normalised default stylesheet shipped with this package."""
|
||||
data = resources.files("mews.data").joinpath("mews-0.1.css").read_text("utf-8")
|
||||
return normalise(data)
|
||||
|
||||
|
||||
def normalise(text: str) -> str:
|
||||
"""Strip the differences a copy may pick up without changing any rule."""
|
||||
text = text.lstrip("").replace("\r\n", "\n").replace("\r", "\n")
|
||||
lines = [line.rstrip() for line in text.split("\n")]
|
||||
out: list[str] = []
|
||||
for line in lines:
|
||||
if line or (out and out[-1]):
|
||||
out.append(line)
|
||||
return "\n".join(out).strip("\n")
|
||||
|
||||
|
||||
def strip_font_faces(text: str) -> tuple[str, list[str]]:
|
||||
"""Remove top-level @font-face blocks, returning the rest and the blocks.
|
||||
|
||||
Only brace depth zero counts, so an @font-face inside a media query is left
|
||||
in place and shows up as a difference.
|
||||
"""
|
||||
blocks: list[str] = []
|
||||
out: list[str] = []
|
||||
index = 0
|
||||
depth = 0
|
||||
while index < len(text):
|
||||
if depth == 0 and FONT_FACE.match(text, index):
|
||||
end = _block_end(text, index)
|
||||
if end is None:
|
||||
break
|
||||
blocks.append(text[index:end])
|
||||
index = end
|
||||
continue
|
||||
character = text[index]
|
||||
if character == "{":
|
||||
depth += 1
|
||||
elif character == "}":
|
||||
depth = max(0, depth - 1)
|
||||
out.append(character)
|
||||
index += 1
|
||||
out.append(text[index:])
|
||||
return "".join(out), blocks
|
||||
|
||||
|
||||
def _block_end(text: str, start: int) -> int | None:
|
||||
"""Index just past the brace-balanced block beginning at start."""
|
||||
opened = text.find("{", start)
|
||||
if opened == -1:
|
||||
return None
|
||||
depth = 0
|
||||
for index in range(opened, len(text)):
|
||||
if text[index] == "{":
|
||||
depth += 1
|
||||
elif text[index] == "}":
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
return index + 1
|
||||
return None
|
||||
|
||||
|
||||
def compare(candidate: str) -> Comparison:
|
||||
"""Compare a stylesheet against the default, allowing added @font-face rules."""
|
||||
expected = canonical()
|
||||
if normalise(candidate) == expected:
|
||||
return Comparison(equal=True)
|
||||
|
||||
remainder, blocks = strip_font_faces(candidate)
|
||||
found = normalise(remainder)
|
||||
has_import = bool(IMPORT_RULE.search(candidate))
|
||||
if found == expected:
|
||||
return Comparison(equal=True, font_faces=blocks, has_import=has_import)
|
||||
|
||||
diff = list(
|
||||
difflib.unified_diff(
|
||||
expected.split("\n"), found.split("\n"), "mews-0.1.css", "your copy", n=1
|
||||
)
|
||||
)
|
||||
line = next(
|
||||
(
|
||||
index + 1
|
||||
for index, (left, right) in enumerate(
|
||||
zip(expected.split("\n"), found.split("\n"), strict=False)
|
||||
)
|
||||
if left != right
|
||||
),
|
||||
min(len(expected.split("\n")), len(found.split("\n"))) + 1,
|
||||
)
|
||||
return Comparison(
|
||||
equal=False,
|
||||
diff_line=line,
|
||||
diff="\n".join(diff[:8]),
|
||||
font_faces=blocks,
|
||||
has_import=has_import,
|
||||
)
|
||||
7
mews/data/README.md
Normal file
7
mews/data/README.md
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
# Packaged copies
|
||||
|
||||
The canonical files live at the repository root (`mews-0.1.css`) and in `dtd/`
|
||||
(`mews-0.1.dtd`), where SPEC.md and `dtd/check_dtd.py` reference them. These
|
||||
copies exist so the validator still finds them when it is installed as a wheel,
|
||||
with nothing but the package on disk. `tests/test_data.py` fails if a copy and
|
||||
its original drift apart.
|
||||
106
mews/data/mews-0.1.css
Normal file
106
mews/data/mews-0.1.css
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
/* Mews Profile default stylesheet 0.1 */
|
||||
|
||||
/* Base: light theme */
|
||||
body {
|
||||
background-color: #faf8f3;
|
||||
color: #1f1f1f;
|
||||
font-family: "Atkinson Hyperlegible", Verdana, Tahoma, system-ui, sans-serif;
|
||||
font-size: 106%;
|
||||
line-height: 1.5;
|
||||
margin: 0 auto;
|
||||
padding: 1em;
|
||||
max-width: 40em;
|
||||
}
|
||||
|
||||
h1, h2, h3, h4, h5, h6 {
|
||||
font-weight: bold;
|
||||
line-height: 1.25;
|
||||
margin: 1.6em 0 0.5em 0;
|
||||
}
|
||||
h1 { font-size: 1.6em; margin-top: 0.5em; }
|
||||
h2 { font-size: 1.3em; }
|
||||
h3 { font-size: 1.1em; }
|
||||
h4, h5, h6 { font-size: 1em; }
|
||||
|
||||
p, ul, ol, dl, blockquote, pre, table, address {
|
||||
margin: 0 0 1em 0;
|
||||
}
|
||||
ul, ol { padding-left: 1.5em; }
|
||||
li { margin-bottom: 0.25em; }
|
||||
dt { font-weight: bold; }
|
||||
dd { margin: 0 0 0.5em 1.5em; }
|
||||
|
||||
a { color: #1a4f9c; text-decoration: underline; }
|
||||
a:visited { color: #5b2a86; }
|
||||
a:focus, a:active { outline: 2px solid #1a4f9c; }
|
||||
|
||||
blockquote {
|
||||
margin-left: 0;
|
||||
padding-left: 1em;
|
||||
border-left: 3px solid #c9c4b8;
|
||||
}
|
||||
|
||||
pre, code, kbd, samp {
|
||||
font-family: "Source Code Pro", Consolas, Menlo, "DejaVu Sans Mono", monospace;
|
||||
font-size: 0.9em;
|
||||
}
|
||||
pre {
|
||||
background-color: #efece4;
|
||||
padding: 0.75em;
|
||||
overflow: auto;
|
||||
}
|
||||
|
||||
hr {
|
||||
border: 0;
|
||||
border-top: 1px solid #c9c4b8;
|
||||
margin: 2em 0;
|
||||
}
|
||||
|
||||
table { border-collapse: collapse; }
|
||||
th, td {
|
||||
border: 1px solid #c9c4b8;
|
||||
padding: 0.3em 0.6em;
|
||||
text-align: left;
|
||||
vertical-align: top;
|
||||
}
|
||||
caption { font-weight: bold; text-align: left; }
|
||||
|
||||
img {
|
||||
display: block;
|
||||
max-width: 100%;
|
||||
height: auto;
|
||||
border: 0;
|
||||
}
|
||||
|
||||
input, select, textarea { font-size: 1em; font-family: inherit; }
|
||||
|
||||
/* Body variants: light */
|
||||
body.mews-warm { background-color: #fbf4e8; color: #2a2420; }
|
||||
body.mews-warm a { color: #8a3d14; }
|
||||
body.mews-cool { background-color: #f3f6fa; color: #1c2430; }
|
||||
body.mews-cool a { color: #1a4f9c; }
|
||||
body.mews-green { background-color: #f2f6ef; color: #1f2a1f; }
|
||||
body.mews-green a { color: #24502a; }
|
||||
body.mews-mono { background-color: #e8ebe1; color: #1e2219; }
|
||||
body.mews-mono a { color: #1e2219; }
|
||||
|
||||
/* Dark theme: follows the system setting */
|
||||
@media (prefers-color-scheme: dark) {
|
||||
body { background-color: #18191b; color: #e3e1dc; }
|
||||
a { color: #8fb8f5; }
|
||||
a:visited { color: #c9a3f2; }
|
||||
a:focus, a:active { outline-color: #8fb8f5; }
|
||||
blockquote, hr, th, td { border-color: #45464a; }
|
||||
pre { background-color: #222326; }
|
||||
|
||||
body.mews-warm { background-color: #1d1916; color: #e8dfd3; }
|
||||
body.mews-warm a { color: #f0b48a; }
|
||||
body.mews-cool { background-color: #161a20; color: #dde4ee; }
|
||||
body.mews-cool a { color: #8fb8f5; }
|
||||
body.mews-green { background-color: #161b16; color: #dbe6d8; }
|
||||
body.mews-green a { color: #9fd39a; }
|
||||
body.mews-mono { background-color: #1b1d18; color: #c8d0b8; }
|
||||
body.mews-mono a { color: #c8d0b8; }
|
||||
}
|
||||
|
||||
html { color-scheme: light dark; }
|
||||
396
mews/data/mews-0.1.dtd
Normal file
396
mews/data/mews-0.1.dtd
Normal file
|
|
@ -0,0 +1,396 @@
|
|||
<!-- Mews Profile 0.1 DTD
|
||||
Published at mews.page/dtd/mews-0.1.dtd. Browsers and clients must not
|
||||
fetch this file.
|
||||
|
||||
XHTML Mobile Profile 1.2 restricted as in the Mews Profile
|
||||
Specification 0.1 section 4, extended with lang and dir (section 3).
|
||||
Maintained by hand and checked against the XHTML Mobile 1.2 driver,
|
||||
xhtml-mobile12.dtd (OMA-SUP-DTD_xhtml_mobile12-V1_2-20080331-A), by
|
||||
check_dtd.py. It cannot be generated by stripping the driver: the
|
||||
driver declares no elements itself, attribute lists only ever add,
|
||||
and the rules below do not exist in XHTML-MP to be kept.
|
||||
|
||||
This file is standalone: it declares every element and attribute
|
||||
itself, so validation never depends on the Open Mobile Alliance's or
|
||||
W3C's servers. Pages keep the XHTML-MP doctype (section 3.1); only
|
||||
validators use this file (section 9.1).
|
||||
|
||||
Encoded here: the permitted elements and attributes (4.1), the head
|
||||
rules (3.3), the body variant classes (5.3) and the ban on nested
|
||||
tables (4.2). 51 elements are kept from XHTML Mobile 1.2. Dropped:
|
||||
acronym, base, big, button, div, legend, noscript, object, param,
|
||||
script, small, span, style, sub, sup, tt. Event attributes and the
|
||||
style attribute are omitted entirely.
|
||||
-->
|
||||
|
||||
<!-- Datatypes (as in XHTML Modularization) -->
|
||||
<!ENTITY % Text.datatype "CDATA" >
|
||||
<!ENTITY % URI.datatype "CDATA" >
|
||||
<!ENTITY % Number.datatype "CDATA" >
|
||||
<!ENTITY % LanguageCode.datatype "NMTOKEN" >
|
||||
<!ENTITY % Character.datatype "CDATA" >
|
||||
|
||||
<!-- Common attributes. id, title, xml:lang, lang and dir are permitted
|
||||
on every element in the body (SPEC.md 4.1); lang and dir are the Mews
|
||||
extensions over XHTML-MP (SPEC.md 3). -->
|
||||
<!ENTITY % id.attrib "id ID #IMPLIED" >
|
||||
<!ENTITY % title.attrib "title CDATA #IMPLIED" >
|
||||
<!ENTITY % i18n.attrib
|
||||
"xml:lang %LanguageCode.datatype; #IMPLIED
|
||||
lang %LanguageCode.datatype; #IMPLIED
|
||||
dir (ltr | rtl) #IMPLIED" >
|
||||
<!ENTITY % Common.attrib "%id.attrib; %title.attrib; %i18n.attrib;" >
|
||||
|
||||
<!-- Inline classes -->
|
||||
<!ENTITY % InlStruct.class "br" >
|
||||
<!ENTITY % InlPhras.class "| em | strong | code | kbd | samp | var | q | cite | abbr | dfn" >
|
||||
<!ENTITY % InlPres.class "| b | i" >
|
||||
<!ENTITY % Anchor.class "| a" >
|
||||
<!ENTITY % InlSpecial.class "| img" >
|
||||
<!ENTITY % InlFormNoLabel.class "| input | select | textarea" >
|
||||
<!ENTITY % InlForm.class "%InlFormNoLabel.class; | label" >
|
||||
<!ENTITY % Inline.extra "" >
|
||||
<!ENTITY % Inline.class "%InlStruct.class;
|
||||
%InlPhras.class;
|
||||
%InlPres.class;
|
||||
%Anchor.class;
|
||||
%InlSpecial.class;
|
||||
%InlForm.class;
|
||||
%Inline.extra;" >
|
||||
<!ENTITY % InlNoAnchor.class "%InlStruct.class;
|
||||
%InlPhras.class;
|
||||
%InlPres.class;
|
||||
%InlSpecial.class;
|
||||
%InlForm.class;
|
||||
%Inline.extra;" >
|
||||
<!ENTITY % InlNoLabel.class "%InlStruct.class;
|
||||
%InlPhras.class;
|
||||
%InlPres.class;
|
||||
%Anchor.class;
|
||||
%InlSpecial.class;
|
||||
%InlFormNoLabel.class;
|
||||
%Inline.extra;" >
|
||||
|
||||
<!-- Block classes -->
|
||||
<!ENTITY % Heading.class "h1 | h2 | h3 | h4 | h5 | h6" >
|
||||
<!ENTITY % List.class "ul | ol | dl" >
|
||||
<!ENTITY % BlkStruct.class "p" >
|
||||
<!ENTITY % BlkPhras.class "| pre | blockquote | address" >
|
||||
<!ENTITY % BlkPres.class "| hr" >
|
||||
<!ENTITY % Table.class "| table" >
|
||||
<!ENTITY % Form.class "| form" >
|
||||
<!ENTITY % Fieldset.class "| fieldset" >
|
||||
<!ENTITY % Block.extra "" >
|
||||
<!ENTITY % BlkNoForm.class "%Heading.class;
|
||||
| %List.class;
|
||||
| %BlkStruct.class;
|
||||
%BlkPhras.class;
|
||||
%BlkPres.class;
|
||||
%Fieldset.class;
|
||||
%Block.extra;" >
|
||||
<!ENTITY % BlkNoTable.class "%BlkNoForm.class;
|
||||
%Form.class;
|
||||
%Block.extra;" >
|
||||
<!ENTITY % Blk.class "%BlkNoTable.class;
|
||||
%Table.class;
|
||||
%Block.extra;" >
|
||||
<!ENTITY % FlowNoTable.class "%Inline.class;
|
||||
| %BlkNoTable.class;" >
|
||||
|
||||
<!-- pre content model, taken verbatim from the driver. It excludes
|
||||
%InlPres.class; and %InlSpecial.class;, so b, i and img are not
|
||||
permitted in pre, as in XHTML Mobile 1.2. -->
|
||||
<!ENTITY % pre.content
|
||||
"( #PCDATA
|
||||
| %InlStruct.class;
|
||||
%InlPhras.class;
|
||||
%Anchor.class;
|
||||
%Inline.extra; )*"
|
||||
>
|
||||
|
||||
<!-- Document structure (SPEC.md 4.1) -->
|
||||
<!--
|
||||
Only body uses %Blk.class;, so a table can appear directly in the
|
||||
body and nowhere else, and tables cannot nest (SPEC.md 4.2). body may
|
||||
carry one class from the variant list (SPEC.md 5.3).
|
||||
-->
|
||||
<!ELEMENT html ( head, body ) >
|
||||
<!ATTLIST html
|
||||
xmlns CDATA #FIXED "http://www.w3.org/1999/xhtml"
|
||||
xml:lang %LanguageCode.datatype; #IMPLIED
|
||||
lang %LanguageCode.datatype; #IMPLIED
|
||||
dir (ltr | rtl) #IMPLIED
|
||||
>
|
||||
<!ELEMENT head ( title, meta, ( meta | link )* ) >
|
||||
<!ELEMENT title ( #PCDATA )* >
|
||||
<!ELEMENT body ( %Blk.class; )+ >
|
||||
<!ATTLIST body
|
||||
%Common.attrib;
|
||||
class (mews-warm | mews-cool | mews-green | mews-mono) #IMPLIED
|
||||
>
|
||||
|
||||
<!-- Head elements (SPEC.md 3.3) -->
|
||||
<!--
|
||||
The head may contain only title, meta and link. The title comes
|
||||
first and at least one meta is required; the conformance marker is a
|
||||
meta (SPEC.md 3.2), but a DTD cannot check its name and content
|
||||
together, so validators confirm the marker as a rule check (SPEC.md
|
||||
9.1). link rel is limited to the uses SPEC.md 3.3 names: the
|
||||
stylesheet, the icon and feed links.
|
||||
-->
|
||||
<!ELEMENT meta EMPTY >
|
||||
<!ATTLIST meta
|
||||
name NMTOKEN #IMPLIED
|
||||
content CDATA #REQUIRED
|
||||
>
|
||||
<!ELEMENT link EMPTY >
|
||||
<!ATTLIST link
|
||||
rel (alternate | icon | stylesheet) #IMPLIED
|
||||
href %URI.datatype; #IMPLIED
|
||||
type CDATA #IMPLIED
|
||||
title CDATA #IMPLIED
|
||||
>
|
||||
|
||||
<!-- Headings (SPEC.md 4.1) -->
|
||||
<!ELEMENT h1 ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST h1
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT h2 ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST h2
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT h3 ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST h3
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT h4 ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST h4
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT h5 ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST h5
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT h6 ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST h6
|
||||
%Common.attrib;
|
||||
>
|
||||
|
||||
<!-- Text blocks (SPEC.md 4.1) -->
|
||||
<!ELEMENT p ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST p
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT blockquote ( %BlkNoTable.class; )* >
|
||||
<!ATTLIST blockquote
|
||||
%Common.attrib;
|
||||
cite %URI.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT pre %pre.content; >
|
||||
<!ATTLIST pre
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT address ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST address
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT hr EMPTY >
|
||||
<!ATTLIST hr
|
||||
%Common.attrib;
|
||||
>
|
||||
|
||||
<!-- Inline text (SPEC.md 4.1) -->
|
||||
<!ELEMENT em ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST em
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT strong ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST strong
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT code ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST code
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT kbd ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST kbd
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT samp ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST samp
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT var ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST var
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT cite ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST cite
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT abbr ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST abbr
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT dfn ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST dfn
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT b ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST b
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT i ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST i
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT q ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST q
|
||||
%Common.attrib;
|
||||
cite %URI.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT br EMPTY >
|
||||
<!ATTLIST br
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT a ( #PCDATA | %InlNoAnchor.class; )* >
|
||||
<!ATTLIST a
|
||||
%Common.attrib;
|
||||
href %URI.datatype; #IMPLIED
|
||||
hreflang %LanguageCode.datatype; #IMPLIED
|
||||
rel CDATA #IMPLIED
|
||||
type CDATA #IMPLIED
|
||||
accesskey %Character.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT img EMPTY >
|
||||
<!ATTLIST img
|
||||
%Common.attrib;
|
||||
src %URI.datatype; #REQUIRED
|
||||
alt %Text.datatype; #REQUIRED
|
||||
>
|
||||
|
||||
<!-- Lists (SPEC.md 4.1) -->
|
||||
<!ELEMENT ul ( li )+ >
|
||||
<!ATTLIST ul
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT ol ( li )+ >
|
||||
<!ATTLIST ol
|
||||
%Common.attrib;
|
||||
start %Number.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT dl ( dt | dd )+ >
|
||||
<!ATTLIST dl
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT li ( #PCDATA | %FlowNoTable.class; )* >
|
||||
<!ATTLIST li
|
||||
%Common.attrib;
|
||||
value %Number.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT dt ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST dt
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT dd ( #PCDATA | %FlowNoTable.class; )* >
|
||||
<!ATTLIST dd
|
||||
%Common.attrib;
|
||||
>
|
||||
|
||||
<!-- Tables (SPEC.md 4.2) -->
|
||||
<!--
|
||||
As in the XHTML Mobile basic tables module, cells take
|
||||
%FlowNoTable.class; and never a table, so tables cannot nest
|
||||
(SPEC.md 4.2).
|
||||
-->
|
||||
<!ELEMENT table ( caption?, tr+ ) >
|
||||
<!ATTLIST table
|
||||
%Common.attrib;
|
||||
summary %Text.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT caption ( #PCDATA | %Inline.class; )* >
|
||||
<!ATTLIST caption
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT tr ( th | td )+ >
|
||||
<!ATTLIST tr
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT th ( #PCDATA | %FlowNoTable.class; )* >
|
||||
<!ATTLIST th
|
||||
%Common.attrib;
|
||||
rowspan %Number.datatype; "1"
|
||||
colspan %Number.datatype; "1"
|
||||
scope (row | col) #IMPLIED
|
||||
>
|
||||
<!ELEMENT td ( #PCDATA | %FlowNoTable.class; )* >
|
||||
<!ATTLIST td
|
||||
%Common.attrib;
|
||||
rowspan %Number.datatype; "1"
|
||||
colspan %Number.datatype; "1"
|
||||
scope (row | col) #IMPLIED
|
||||
>
|
||||
|
||||
<!-- Forms (SPEC.md 4.1, 4.2) -->
|
||||
<!--
|
||||
As in the XHTML MP forms module, form takes %BlkNoForm.class;, so
|
||||
forms cannot nest. fieldset also takes %BlkNoForm.class;, so a
|
||||
fieldset cannot reintroduce a form.
|
||||
-->
|
||||
<!ELEMENT form ( %BlkNoForm.class; )+ >
|
||||
<!ATTLIST form
|
||||
%Common.attrib;
|
||||
action %URI.datatype; #REQUIRED
|
||||
method (get | post) "get"
|
||||
enctype CDATA "application/x-www-form-urlencoded"
|
||||
>
|
||||
<!ELEMENT fieldset ( #PCDATA | %Inline.class; | %BlkNoForm.class; )* >
|
||||
<!ATTLIST fieldset
|
||||
%Common.attrib;
|
||||
>
|
||||
<!ELEMENT label ( #PCDATA | %InlNoLabel.class; )* >
|
||||
<!ATTLIST label
|
||||
%Common.attrib;
|
||||
for IDREF #IMPLIED
|
||||
accesskey %Character.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT input EMPTY >
|
||||
<!ATTLIST input
|
||||
%Common.attrib;
|
||||
type (text | password | checkbox | radio | submit | reset | hidden) "text"
|
||||
name CDATA #IMPLIED
|
||||
value CDATA #IMPLIED
|
||||
checked (checked) #IMPLIED
|
||||
size %Number.datatype; #IMPLIED
|
||||
maxlength %Number.datatype; #IMPLIED
|
||||
accesskey %Character.datatype; #IMPLIED
|
||||
inputmode CDATA #IMPLIED
|
||||
>
|
||||
<!ELEMENT select ( optgroup | option )+ >
|
||||
<!ATTLIST select
|
||||
%Common.attrib;
|
||||
name CDATA #IMPLIED
|
||||
multiple (multiple) #IMPLIED
|
||||
size %Number.datatype; #IMPLIED
|
||||
>
|
||||
<!ELEMENT optgroup ( option )+ >
|
||||
<!ATTLIST optgroup
|
||||
%Common.attrib;
|
||||
label %Text.datatype; #REQUIRED
|
||||
>
|
||||
<!ELEMENT option ( #PCDATA )* >
|
||||
<!ATTLIST option
|
||||
%Common.attrib;
|
||||
value CDATA #IMPLIED
|
||||
selected (selected) #IMPLIED
|
||||
>
|
||||
<!ELEMENT textarea ( #PCDATA )* >
|
||||
<!ATTLIST textarea
|
||||
%Common.attrib;
|
||||
name CDATA #IMPLIED
|
||||
rows %Number.datatype; #REQUIRED
|
||||
cols %Number.datatype; #REQUIRED
|
||||
accesskey %Character.datatype; #IMPLIED
|
||||
inputmode CDATA #IMPLIED
|
||||
>
|
||||
382
mews/fetch.py
Normal file
382
mews/fetch.py
Normal file
|
|
@ -0,0 +1,382 @@
|
|||
"""Fetching pages that strangers ask the checker to look at.
|
||||
|
||||
Every URL reaching this module is author-supplied, so the service must not be
|
||||
usable as a probe against the machine it runs on or its neighbours. Addresses
|
||||
are vetted before the connection and the connection is pinned to the vetted
|
||||
address, so a second DNS answer cannot redirect it (DNS rebinding).
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass, replace
|
||||
import ipaddress
|
||||
import os
|
||||
import re
|
||||
import socket
|
||||
import time
|
||||
from urllib.parse import urlsplit, urlunsplit
|
||||
|
||||
import httpx
|
||||
from publicsuffixlist import PublicSuffixList
|
||||
|
||||
# Read caps, in bytes. The page cap is above the 256 KB that SPEC.md 4.3 makes a
|
||||
# MUST so an oversized page is reported as too large rather than unreachable.
|
||||
PAGE_CAP = 288 * 1024
|
||||
IMAGE_CAP = 64 * 1024
|
||||
CSS_CAP = 256 * 1024
|
||||
|
||||
MAX_REDIRECTS = 3
|
||||
MAX_URL_LENGTH = 2048
|
||||
USER_AGENT = "mews.page checker (+https://mews.page/about)"
|
||||
|
||||
_PSL = PublicSuffixList()
|
||||
|
||||
# Ranges ipaddress does not classify but which must never be reached: carrier
|
||||
# NAT (which is also Tailscale's range), IETF protocol assignment, benchmarking,
|
||||
# the documentation ranges, and the IPv6 transition mechanisms that tunnel an
|
||||
# IPv4 address inside an address that looks global.
|
||||
_BLOCKED_NETS = [
|
||||
ipaddress.ip_network(net)
|
||||
for net in (
|
||||
"100.64.0.0/10",
|
||||
"192.0.0.0/24",
|
||||
"198.18.0.0/15",
|
||||
"192.0.2.0/24",
|
||||
"198.51.100.0/24",
|
||||
"203.0.113.0/24",
|
||||
"240.0.0.0/4",
|
||||
"255.255.255.255/32",
|
||||
"2001:db8::/32",
|
||||
"2002::/16",
|
||||
"2001::/32",
|
||||
"64:ff9b::/96",
|
||||
)
|
||||
]
|
||||
|
||||
# Hosts whose own addresses must be unreachable, as CIDRs. The deployment sets
|
||||
# this to the machine's public address so a submission cannot be aimed at a
|
||||
# service sharing the box.
|
||||
_EXTRA_NETS = [
|
||||
ipaddress.ip_network(net.strip())
|
||||
for net in os.environ.get("MEWS_BLOCK_NETS", "").split(",")
|
||||
if net.strip()
|
||||
]
|
||||
|
||||
_LOCAL_SUFFIXES = (".local", ".internal", ".home.arpa", ".localhost")
|
||||
|
||||
# What a host name may be made of. Checking this first means a typo gets a
|
||||
# clearer answer than a complaint about public suffixes.
|
||||
_HOSTNAME = re.compile(
|
||||
r"^[a-z0-9]([a-z0-9-]*[a-z0-9])?(\.[a-z0-9]([a-z0-9-]*[a-z0-9])?)*$"
|
||||
)
|
||||
|
||||
|
||||
class UrlError(Exception):
|
||||
"""A URL the checker will not fetch. The message is shown to the author."""
|
||||
|
||||
|
||||
class FetchError(Exception):
|
||||
"""A fetch that produced no page. The message is shown to the author."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Fetched:
|
||||
"""One response the checker read, with the body it kept."""
|
||||
|
||||
url: str
|
||||
status: int
|
||||
headers: httpx.Headers
|
||||
body: bytes
|
||||
truncated: bool
|
||||
scheme_downgraded: bool = False
|
||||
|
||||
|
||||
def registered_domain(host: str) -> str | None:
|
||||
"""Return the registrable domain of host, or None when it has no public suffix.
|
||||
|
||||
Comparing registrable domains rather than hostnames is what makes the
|
||||
same-site rules in SPEC.md 4.2 and 7.4 mean anything: a free subdomain host
|
||||
would otherwise let any two unrelated sites count as one.
|
||||
"""
|
||||
if not host:
|
||||
return None
|
||||
try:
|
||||
name = host.strip().rstrip(".").lower().encode("idna").decode("ascii")
|
||||
except (UnicodeError, UnicodeDecodeError):
|
||||
return None
|
||||
return _PSL.privatesuffix(name)
|
||||
|
||||
|
||||
def same_site(a: str, b: str) -> bool:
|
||||
"""Whether two hosts share a registrable domain."""
|
||||
first = registered_domain(a)
|
||||
return first is not None and first == registered_domain(b)
|
||||
|
||||
|
||||
def normalise_url(raw: str, *, allow_loopback: bool = False) -> str:
|
||||
"""Return a URL the checker is willing to fetch, or raise UrlError.
|
||||
|
||||
This runs before any name lookup, so a rejected URL costs nothing. With
|
||||
allow_loopback the rules relax enough to reach a test server on this
|
||||
machine: an address literal, localhost, and any port.
|
||||
"""
|
||||
text = (raw or "").strip()
|
||||
if not text:
|
||||
raise UrlError("Enter the address of a page to check.")
|
||||
if len(text) > MAX_URL_LENGTH:
|
||||
raise UrlError("That address is too long.")
|
||||
if "://" not in text:
|
||||
text = "https://" + text
|
||||
|
||||
parts = urlsplit(text)
|
||||
if parts.scheme not in ("http", "https"):
|
||||
raise UrlError("Enter an address that starts with http:// or https://.")
|
||||
if "@" in parts.netloc:
|
||||
raise UrlError("Enter an address without a user name in it.")
|
||||
|
||||
try:
|
||||
host, port = parts.hostname, parts.port
|
||||
except ValueError as error:
|
||||
raise UrlError("That address has a port the checker can't read.") from error
|
||||
if not host:
|
||||
raise UrlError(
|
||||
"That doesn't look like a web address. Enter the full address "
|
||||
"of a page, like https://example.com/"
|
||||
)
|
||||
if port is not None and port not in (80, 443) and not allow_loopback:
|
||||
raise UrlError("The checker reads pages on the usual web ports only.")
|
||||
|
||||
host = host.lower().rstrip(".")
|
||||
if _is_ip_literal(host):
|
||||
if not (allow_loopback and _is_loopback_literal(host)):
|
||||
raise UrlError("Enter a domain name rather than an IP address.")
|
||||
elif host == "localhost" or host.endswith(_LOCAL_SUFFIXES):
|
||||
if not allow_loopback:
|
||||
raise UrlError("That address is only reachable on a local network.")
|
||||
elif not _HOSTNAME.match(_ascii(host)):
|
||||
raise UrlError(
|
||||
"That doesn't look like a web address. Enter the full address "
|
||||
"of a page, like https://example.com/"
|
||||
)
|
||||
elif registered_domain(host) is None:
|
||||
raise UrlError("That domain name isn't one the checker can reach.")
|
||||
|
||||
netloc = host if port is None else f"{host}:{port}"
|
||||
return urlunsplit((parts.scheme, netloc, parts.path or "/", parts.query, ""))
|
||||
|
||||
|
||||
def _ascii(host: str) -> str:
|
||||
"""Return the punycode form of a host name, or the name unchanged."""
|
||||
try:
|
||||
return host.encode("idna").decode("ascii")
|
||||
except (UnicodeError, UnicodeDecodeError):
|
||||
return host
|
||||
|
||||
|
||||
def _is_ip_literal(host: str) -> bool:
|
||||
"""Whether host is an address literal in any of the forms a parser accepts."""
|
||||
candidate = host.strip("[]")
|
||||
try:
|
||||
ipaddress.ip_address(candidate)
|
||||
except ValueError:
|
||||
pass
|
||||
else:
|
||||
return True
|
||||
# Decimal, octal and hex integer forms of an IPv4 address, which urlsplit
|
||||
# leaves alone but a resolver would accept.
|
||||
if host.isdigit():
|
||||
return True
|
||||
bare = host.replace(".", "")
|
||||
return bare.startswith(("0x", "0X")) or (
|
||||
host.startswith("0") and len(host) > 1 and bare.isdigit()
|
||||
)
|
||||
|
||||
|
||||
def _is_loopback_literal(host: str) -> bool:
|
||||
"""Whether host is a literal address on this machine."""
|
||||
try:
|
||||
return ipaddress.ip_address(host.strip("[]")).is_loopback
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
def _embedded(address: ipaddress.IPv6Address) -> ipaddress.IPv4Address | None:
|
||||
"""Return the IPv4 address a transition mechanism hides inside an IPv6 one."""
|
||||
for attribute in ("ipv4_mapped", "sixtofour"):
|
||||
value = getattr(address, attribute, None)
|
||||
if value is not None:
|
||||
return value
|
||||
teredo = getattr(address, "teredo", None)
|
||||
if teredo:
|
||||
return teredo[1]
|
||||
if address in ipaddress.ip_network("64:ff9b::/96"):
|
||||
return ipaddress.IPv4Address(int(address) & 0xFFFFFFFF)
|
||||
return None
|
||||
|
||||
|
||||
def vet_address(address: str, *, allow_loopback: bool = False) -> None:
|
||||
"""Raise UrlError unless address is a public one the checker may connect to."""
|
||||
ip = ipaddress.ip_address(address)
|
||||
if allow_loopback and ip.is_loopback:
|
||||
return
|
||||
if (
|
||||
ip.is_private
|
||||
or ip.is_loopback
|
||||
or ip.is_link_local
|
||||
or ip.is_reserved
|
||||
or ip.is_multicast
|
||||
or ip.is_unspecified
|
||||
or any(ip in net for net in _BLOCKED_NETS)
|
||||
or any(ip in net for net in _EXTRA_NETS)
|
||||
):
|
||||
raise UrlError("That address isn't on the public internet.")
|
||||
if isinstance(ip, ipaddress.IPv6Address):
|
||||
inner = _embedded(ip)
|
||||
if inner is not None:
|
||||
vet_address(str(inner), allow_loopback=allow_loopback)
|
||||
|
||||
|
||||
def resolve(url: str, *, allow_loopback: bool = False) -> str:
|
||||
"""Return one vetted address for the URL's host, rejecting the host if any fails.
|
||||
|
||||
Every answer has to pass: a host that resolves to one public and one private
|
||||
address would otherwise be a coin toss.
|
||||
"""
|
||||
parts = urlsplit(url)
|
||||
host = parts.hostname or ""
|
||||
port = parts.port or (443 if parts.scheme == "https" else 80)
|
||||
try:
|
||||
infos = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM)
|
||||
except socket.gaierror as error:
|
||||
raise FetchError(
|
||||
"We couldn't find that domain name. Check the address and try again."
|
||||
) from error
|
||||
addresses = [info[4][0] for info in infos]
|
||||
if not addresses:
|
||||
raise FetchError("We couldn't find that domain name.")
|
||||
for address in addresses:
|
||||
vet_address(address, allow_loopback=allow_loopback)
|
||||
return addresses[0]
|
||||
|
||||
|
||||
class Fetcher:
|
||||
"""Reads pages and their sub-resources, under a time and byte budget.
|
||||
|
||||
One Fetcher serves one check, so the whole-check budget is shared across the
|
||||
page, its stylesheet and its images.
|
||||
"""
|
||||
|
||||
def __init__(self, *, allow_loopback: bool = False, budget: float = 45.0):
|
||||
self.allow_loopback = allow_loopback
|
||||
self._deadline = time.monotonic() + budget
|
||||
self._client = httpx.Client(
|
||||
follow_redirects=False,
|
||||
trust_env=False,
|
||||
http2=False,
|
||||
verify=True,
|
||||
timeout=httpx.Timeout(connect=5.0, read=10.0, write=5.0, pool=5.0),
|
||||
limits=httpx.Limits(max_connections=4, max_keepalive_connections=2),
|
||||
headers={
|
||||
"User-Agent": USER_AGENT,
|
||||
"Accept": "text/html,application/xhtml+xml",
|
||||
"Accept-Encoding": "gzip",
|
||||
},
|
||||
)
|
||||
|
||||
def __enter__(self) -> "Fetcher":
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc: object) -> None:
|
||||
self.close()
|
||||
|
||||
def close(self) -> None:
|
||||
"""Close the underlying connections."""
|
||||
self._client.close()
|
||||
|
||||
def get(
|
||||
self,
|
||||
url: str,
|
||||
*,
|
||||
cap: int = PAGE_CAP,
|
||||
headers: dict[str, str] | None = None,
|
||||
) -> Fetched:
|
||||
"""Fetch one URL, following redirects by hand and vetting each hop."""
|
||||
current = normalise_url(url, allow_loopback=self.allow_loopback)
|
||||
downgraded = False
|
||||
for _ in range(MAX_REDIRECTS + 1):
|
||||
response = self._request(current, cap=cap, headers=headers)
|
||||
if response.status not in (301, 302, 303, 307, 308):
|
||||
if downgraded:
|
||||
return replace(response, scheme_downgraded=True)
|
||||
return response
|
||||
location = response.headers.get("location", "")
|
||||
if not location:
|
||||
raise FetchError("That page redirects without saying where to.")
|
||||
target = normalise_url(
|
||||
str(httpx.URL(current).join(location)),
|
||||
allow_loopback=self.allow_loopback,
|
||||
)
|
||||
was_secure = urlsplit(current).scheme == "https"
|
||||
if was_secure and urlsplit(target).scheme == "http":
|
||||
downgraded = True
|
||||
current = target
|
||||
raise FetchError("That page redirects too many times.")
|
||||
|
||||
def _request(
|
||||
self,
|
||||
url: str,
|
||||
*,
|
||||
cap: int,
|
||||
headers: dict[str, str] | None,
|
||||
) -> Fetched:
|
||||
"""Make one pinned request, streaming the body up to cap bytes."""
|
||||
self._check_deadline()
|
||||
parts = urlsplit(url)
|
||||
host = parts.hostname or ""
|
||||
port = parts.port
|
||||
address = resolve(url, allow_loopback=self.allow_loopback)
|
||||
literal = f"[{address}]" if ":" in address else address
|
||||
pinned = urlunsplit(
|
||||
(
|
||||
parts.scheme,
|
||||
literal if port is None else f"{literal}:{port}",
|
||||
parts.path,
|
||||
parts.query,
|
||||
"",
|
||||
)
|
||||
)
|
||||
request_headers = {"Host": parts.netloc, **(headers or {})}
|
||||
extensions = {"sni_hostname": host} if parts.scheme == "https" else {}
|
||||
try:
|
||||
with self._client.stream(
|
||||
"GET", pinned, headers=request_headers, extensions=extensions
|
||||
) as response:
|
||||
declared = response.headers.get("content-length")
|
||||
if declared and declared.isdigit() and int(declared) > cap:
|
||||
raise FetchError("That page is too large for the checker to read.")
|
||||
body, truncated = self._read(response, cap)
|
||||
except httpx.TimeoutException as error:
|
||||
raise FetchError("That page took too long to answer.") from error
|
||||
except httpx.HTTPError as error:
|
||||
raise FetchError("We couldn't connect to that site.") from error
|
||||
return Fetched(url, response.status_code, response.headers, body, truncated)
|
||||
|
||||
def _read(self, response: httpx.Response, cap: int) -> tuple[bytes, bool]:
|
||||
"""Read a streamed body, stopping at cap decoded bytes."""
|
||||
chunks: list[bytes] = []
|
||||
total = 0
|
||||
for chunk in response.iter_bytes():
|
||||
self._check_deadline()
|
||||
chunks.append(chunk)
|
||||
total += len(chunk)
|
||||
if total > cap:
|
||||
return b"".join(chunks)[:cap], True
|
||||
body = b"".join(chunks)
|
||||
encoded = response.num_bytes_downloaded or len(body)
|
||||
# A body that expanded enormously from a small download is a compression
|
||||
# bomb, not a page.
|
||||
if encoded and len(body) > 100 * encoded and len(body) > 64 * 1024:
|
||||
raise FetchError("That page is compressed in a way the checker won't read.")
|
||||
return body, False
|
||||
|
||||
def _check_deadline(self) -> None:
|
||||
if time.monotonic() > self._deadline:
|
||||
raise FetchError("Checking that page took too long.")
|
||||
976
mews/lint.py
Normal file
976
mews/lint.py
Normal file
|
|
@ -0,0 +1,976 @@
|
|||
"""The Mews page validator.
|
||||
|
||||
SPEC.md 9.1 defines conformance as two checks: the page validates against
|
||||
dtd/mews-0.1.dtd, and it follows the rules a DTD cannot express. This module
|
||||
runs both and reports every failure with its spec section.
|
||||
|
||||
Failures at MUST level mean the page does not conform. Failures at SHOULD level
|
||||
are warnings: they are worth fixing but do not make a page non-conforming.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
from dataclasses import dataclass, field
|
||||
import html.entities
|
||||
from importlib import resources
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sys
|
||||
from urllib.parse import urljoin, urlsplit
|
||||
|
||||
from lxml import etree
|
||||
|
||||
from mews import VERSION, css as mews_css
|
||||
from mews.fetch import (
|
||||
CSS_CAP,
|
||||
IMAGE_CAP,
|
||||
Fetched,
|
||||
Fetcher,
|
||||
FetchError,
|
||||
UrlError,
|
||||
normalise_url,
|
||||
same_site,
|
||||
)
|
||||
|
||||
XHTML = "http://www.w3.org/1999/xhtml"
|
||||
XML_LANG = "{http://www.w3.org/XML/1998/namespace}lang"
|
||||
|
||||
MARKER = "mews-profile"
|
||||
CANONICAL_STYLESHEET = "https://mews.page/mews-0.1.css"
|
||||
|
||||
OK = 200
|
||||
MAX_FINDINGS = 50
|
||||
MAX_IMAGES = 20
|
||||
|
||||
SIZE_MUST = 256 * 1024
|
||||
SIZE_SHOULD = 64 * 1024
|
||||
SIZE_TOTAL = 320 * 1024
|
||||
IMAGE_SHOULD = 50 * 1024
|
||||
|
||||
XML_DECLARATION = re.compile(rb'^<\?xml version="1\.0" encoding="(?i:UTF-8)"\?>')
|
||||
DOCTYPE = re.compile(
|
||||
rb'<!DOCTYPE\s+html\s+PUBLIC\s+"-//WAPFORUM//DTD XHTML Mobile 1\.2//EN"\s+'
|
||||
rb'"http://www\.openmobilealliance\.org/tech/DTD/xhtml-mobile12\.dtd"\s*>'
|
||||
)
|
||||
|
||||
ALLOWED_META = ("mews-profile", "description", "author", "viewport")
|
||||
ALLOWED_RELS = ("stylesheet", "icon", "alternate")
|
||||
|
||||
# Magic bytes for the three formats SPEC.md 4.2 recommends.
|
||||
IMAGE_MAGIC = {
|
||||
b"GIF87a": "GIF",
|
||||
b"GIF89a": "GIF",
|
||||
b"\x89PNG\r\n\x1a\n": "PNG",
|
||||
b"\xff\xd8\xff": "JPEG",
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Check:
|
||||
"""One thing the validator looks at, and the spec section it comes from."""
|
||||
|
||||
code: str
|
||||
section: str
|
||||
level: str
|
||||
what: str
|
||||
|
||||
|
||||
CHECKS: tuple[Check, ...] = (
|
||||
Check("xml-declaration", "3.1", "must", "UTF-8 XML declaration, no BOM"),
|
||||
Check("doctype", "3.1", "must", "the XHTML-MP 1.2 doctype, exactly"),
|
||||
Check("internal-subset", "3.1", "must", "no internal DTD subset"),
|
||||
Check("well-formed", "3", "must", "the page parses as XML"),
|
||||
Check("root-element", "3.1", "must", "html root in the XHTML namespace"),
|
||||
Check("lang", "3.1", "should", "xml:lang and lang agree on html"),
|
||||
Check("dtd-element", "4.1", "must", "only elements from 4.1"),
|
||||
Check("dtd-attribute", "4.1", "must", "only attributes from 4.1"),
|
||||
Check("dtd-nesting", "4.1", "must", "elements nest as the DTD allows"),
|
||||
Check("dtd-content", "4.1", "must", "element content follows the DTD"),
|
||||
Check("dtd-value", "4.1", "must", "attribute values from the permitted set"),
|
||||
Check("dtd-required-attribute", "4.1", "must", "required attributes present"),
|
||||
Check("dtd-head", "3.3", "must", "the head holds only what 3.3 permits"),
|
||||
Check("dtd-nested-table", "4.2", "must", "tables are not nested"),
|
||||
Check("dtd-variant", "5.3", "must", "body class is one of the variants"),
|
||||
Check("dtd-styling", "5", "must", "no author styling"),
|
||||
Check("dtd-other", "4.1", "must", "the page matches the Mews DTD"),
|
||||
Check("marker", "3.2", "must", "the conformance marker is present once"),
|
||||
Check("marker-version", "3.2", "must", "the marker names a known version"),
|
||||
Check("meta-name", "3.3", "must", "only the meta names 3.3 permits"),
|
||||
Check("link-rel", "3.3", "must", "only the link kinds 3.3 permits"),
|
||||
Check("link-stylesheet-count", "3.3", "must", "at most one stylesheet link"),
|
||||
Check("link-alternate-type", "3.3", "must", "feed links are Atom"),
|
||||
Check("icon-offsite", "3.3", "must", "the icon is on the page's own site"),
|
||||
Check("viewport", "3.3", "should", "the viewport meta is present"),
|
||||
Check("size-must", "4.3", "must", "markup under 256 KB"),
|
||||
Check("size-should", "4.3", "should", "markup under 64 KB"),
|
||||
Check("size-total", "4.3", "should", "page plus images under 320 KB"),
|
||||
Check("image-data-uri", "4.2", "must", "no data: URI images"),
|
||||
Check("image-offsite", "4.2", "must", "images are on the page's own site"),
|
||||
Check("image-format", "4.2", "should", "images are GIF, JPEG or PNG"),
|
||||
Check("image-size", "4.2", "should", "each image under 50 KB"),
|
||||
Check("image-unreachable", "4.2", "should", "images load"),
|
||||
Check("image-metadata", "4.2", "should", "images carry no camera metadata"),
|
||||
Check("images-not-checked", "4.2", "should", "how many images were checked"),
|
||||
Check("stylesheet-missing", "5.2", "should", "the default stylesheet is linked"),
|
||||
Check("stylesheet-canonical", "5.2", "should", "the site hosts its own copy"),
|
||||
Check("stylesheet-modified", "5.2", "must", "the copy is unmodified"),
|
||||
Check("stylesheet-import", "5.2", "must", "the copy has no @import"),
|
||||
Check("stylesheet-unreachable", "5.2", "should", "the stylesheet loads"),
|
||||
Check("font-offsite", "5.2", "must", "@font-face fonts are on the own site"),
|
||||
Check("stylesheet-offsite", "7.4", "must", "the stylesheet is on the own site"),
|
||||
Check("content-type", "7.1", "should", "a page content type"),
|
||||
Check("https", "7.2", "should", "served over HTTPS"),
|
||||
Check("validators", "7.3", "should", "Last-Modified or ETag is sent"),
|
||||
Check("cookie", "7.4", "must", "no cookies are set"),
|
||||
)
|
||||
|
||||
BY_CODE = {check.code: check for check in CHECKS}
|
||||
|
||||
# Rules in sections 3 to 7 this validator deliberately leaves alone. Together
|
||||
# with CHECKS this covers every MUST and SHOULD in those sections; tests fail if
|
||||
# a section appears in neither.
|
||||
NOT_CHECKED = {
|
||||
"5.1": "The default stylesheet is compared byte for byte (5.2), which "
|
||||
"covers whether its base rules are valid WAP CSS.",
|
||||
"6.1": "Dated links are a client convention; the validator reads no feeds.",
|
||||
"6.2": "Atom feed contents are a client convention; only the link type is checked.",
|
||||
"7.2": "Whether a site redirects HTTP to HTTPS in a way old handsets can "
|
||||
"follow cannot be told from one request.",
|
||||
"4.2": "Whether text appears only inside an image cannot be told from markup.",
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Finding:
|
||||
"""One rule a page broke."""
|
||||
|
||||
code: str
|
||||
message: str
|
||||
location: str = ""
|
||||
|
||||
@property
|
||||
def section(self) -> str:
|
||||
"""Return the spec section this finding comes from."""
|
||||
return BY_CODE[self.code].section
|
||||
|
||||
@property
|
||||
def level(self) -> str:
|
||||
"""Return either must or should."""
|
||||
return BY_CODE[self.code].level
|
||||
|
||||
def __str__(self) -> str:
|
||||
where = f" ({self.location})" if self.location else ""
|
||||
return f"Section {self.section} — {self.message}{where}"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Report:
|
||||
"""What the validator found, plus the facts the directory needs."""
|
||||
|
||||
url: str | None = None
|
||||
findings: list[Finding] = field(default_factory=list)
|
||||
notes: list[str] = field(default_factory=list)
|
||||
title: str = ""
|
||||
description: str = ""
|
||||
language: str = ""
|
||||
size: int = 0
|
||||
|
||||
def add(self, code: str, message: str, location: str = "") -> None:
|
||||
"""Record one finding, up to the report cap."""
|
||||
if len(self.findings) < MAX_FINDINGS:
|
||||
self.findings.append(Finding(code, message, location))
|
||||
|
||||
@property
|
||||
def failures(self) -> list[Finding]:
|
||||
"""Return the findings that make the page non-conforming."""
|
||||
return [f for f in self.findings if f.level == "must"]
|
||||
|
||||
@property
|
||||
def warnings(self) -> list[Finding]:
|
||||
"""Return the findings worth fixing that still leave the page conforming."""
|
||||
return [f for f in self.findings if f.level == "should"]
|
||||
|
||||
@property
|
||||
def conforms(self) -> bool:
|
||||
"""Say whether the page follows every MUST rule the validator checks."""
|
||||
return not self.failures
|
||||
|
||||
|
||||
def _entity_declarations() -> str:
|
||||
"""Declare the entity set the XHTML-MP doctype would have defined.
|
||||
|
||||
SPEC.md 9.1 says clients must not fetch the doctype, and the local copy of
|
||||
the driver DTD pulls its modules from w3.org, so the entities are supplied
|
||||
from the stdlib's HTML 4 table instead — the same 252 names.
|
||||
"""
|
||||
return "".join(
|
||||
f'<!ENTITY {name} "&#{code};">'
|
||||
for code, name in sorted(html.entities.codepoint2name.items())
|
||||
)
|
||||
|
||||
|
||||
class _Resolver(etree.Resolver):
|
||||
"""Resolves the XHTML-MP doctype to entity declarations and nothing else."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._entities = _entity_declarations()
|
||||
|
||||
def resolve(self, system_url, public_id, context):
|
||||
"""Answer the parser's request for an external entity."""
|
||||
if public_id == "-//WAPFORUM//DTD XHTML Mobile 1.2//EN":
|
||||
return self.resolve_string(self._entities, context)
|
||||
# With no_network set, returning None makes any other doctype fail.
|
||||
return None
|
||||
|
||||
|
||||
def _parser(*, expand_entities: bool) -> etree.XMLParser:
|
||||
"""Build an XML parser that never reaches the network."""
|
||||
parser = etree.XMLParser(
|
||||
load_dtd=expand_entities,
|
||||
resolve_entities=expand_entities,
|
||||
no_network=True,
|
||||
dtd_validation=False,
|
||||
attribute_defaults=False,
|
||||
huge_tree=False,
|
||||
)
|
||||
if expand_entities:
|
||||
parser.resolvers.add(_Resolver())
|
||||
return parser
|
||||
|
||||
|
||||
_DTD: etree.DTD | None = None
|
||||
|
||||
|
||||
def mews_dtd() -> etree.DTD:
|
||||
"""Return the Mews DTD, loaded once per process."""
|
||||
global _DTD # noqa: PLW0603
|
||||
if _DTD is None:
|
||||
with resources.as_file(
|
||||
resources.files("mews.data").joinpath("mews-0.1.dtd")
|
||||
) as path:
|
||||
_DTD = etree.DTD(str(path))
|
||||
return _DTD
|
||||
|
||||
|
||||
def local(element: etree._Element) -> str:
|
||||
"""Return an element's name without its namespace."""
|
||||
return etree.QName(element).localname
|
||||
|
||||
|
||||
# --- The DTD check -------------------------------------------------------
|
||||
|
||||
UNDECLARED_ELEMENT = re.compile(r"^No declaration for element (\S+)")
|
||||
UNDECLARED_ATTRIBUTE = re.compile(
|
||||
r"^No declaration for attribute (\S+) of element (\S+)"
|
||||
)
|
||||
NOT_ALLOWED_IN = re.compile(
|
||||
r"^Element (\S+) is not declared in (\S+) list of possible children"
|
||||
)
|
||||
CONTENT_MODEL = re.compile(
|
||||
r"^Element (\S+) content does not follow the DTD, expecting (.*?), got \(?(.*?)\)?$"
|
||||
)
|
||||
BAD_VALUE = re.compile(
|
||||
r'^Value "(.*?)" for attribute (\S+) of (\S+) is not among the enumerated set'
|
||||
)
|
||||
MISSING_ATTRIBUTE = re.compile(r"^Element (\S+) does not carry attribute (\S+)")
|
||||
|
||||
|
||||
def _enumerated_values(element: str, attribute: str) -> list[str]:
|
||||
"""Return the values the DTD permits for an attribute, read from the DTD."""
|
||||
for declared in mews_dtd().iterelements():
|
||||
if declared.name != element:
|
||||
continue
|
||||
for attr in declared.iterattributes():
|
||||
if attr.name == attribute:
|
||||
return list(attr.itervalues() or [])
|
||||
return []
|
||||
|
||||
|
||||
def _dtd_findings(report: Report, tree: etree._ElementTree) -> None:
|
||||
"""Validate against the Mews DTD and turn libxml2's errors into findings."""
|
||||
dtd = mews_dtd()
|
||||
if dtd.validate(tree):
|
||||
return
|
||||
|
||||
errors = [(entry.line, entry.message) for entry in dtd.error_log]
|
||||
undeclared = {
|
||||
match.group(1)
|
||||
for _, message in errors
|
||||
for match in [UNDECLARED_ELEMENT.match(message)]
|
||||
if match
|
||||
}
|
||||
|
||||
for line, message in errors:
|
||||
finding = _map_error(message, undeclared)
|
||||
if finding is not None:
|
||||
code, text = finding
|
||||
report.add(code, text, f"line {line}")
|
||||
|
||||
|
||||
def _map_error(message: str, undeclared: set[str]) -> tuple[str, str] | None:
|
||||
"""Map one libxml2 message to a code and a plain sentence, or drop it.
|
||||
|
||||
One mistake makes libxml2 say several things: an unknown element is also an
|
||||
unknown attribute and a content-model break. Only the clearest is kept.
|
||||
"""
|
||||
match = UNDECLARED_ELEMENT.match(message)
|
||||
if match:
|
||||
name = match.group(1)
|
||||
if name == "style":
|
||||
return (
|
||||
"dtd-styling",
|
||||
"The <style> element isn't allowed. Mews pages carry no author "
|
||||
"styling, so readers control how a page looks.",
|
||||
)
|
||||
if name == "script":
|
||||
return (
|
||||
"dtd-element",
|
||||
"The <script> element isn't allowed. Mews clients run no "
|
||||
"scripts, so a page has to work without them.",
|
||||
)
|
||||
return (
|
||||
"dtd-element",
|
||||
f"The <{name}> element isn't allowed. Remove it, or use an element "
|
||||
"from section 4.1.",
|
||||
)
|
||||
|
||||
match = UNDECLARED_ATTRIBUTE.match(message)
|
||||
if match:
|
||||
attribute, element = match.group(1), match.group(2)
|
||||
if element in undeclared:
|
||||
return None
|
||||
if attribute == "style":
|
||||
return (
|
||||
"dtd-styling",
|
||||
f"The style attribute isn't allowed on <{element}>. Mews pages "
|
||||
"carry no author styling.",
|
||||
)
|
||||
return (
|
||||
"dtd-attribute",
|
||||
f"The {attribute} attribute isn't allowed on <{element}>. Remove it.",
|
||||
)
|
||||
|
||||
match = NOT_ALLOWED_IN.match(message)
|
||||
if match:
|
||||
child, parent = match.group(1), match.group(2)
|
||||
if child in undeclared:
|
||||
return None
|
||||
if child == "table":
|
||||
return (
|
||||
"dtd-nested-table",
|
||||
"Tables can't be nested. Move the inner table out, or use a list.",
|
||||
)
|
||||
return ("dtd-nesting", f"<{child}> can't go inside <{parent}>.")
|
||||
|
||||
match = CONTENT_MODEL.match(message)
|
||||
if match:
|
||||
element, got = match.group(1), match.group(3).split()
|
||||
if any(name in undeclared for name in got):
|
||||
return None
|
||||
if element == "head":
|
||||
return (
|
||||
"dtd-head",
|
||||
"The head can hold only a title, then meta and link elements, "
|
||||
"with the title first.",
|
||||
)
|
||||
if not got:
|
||||
if element == "body":
|
||||
return ("dtd-content", "The page has nothing in its body.")
|
||||
return (
|
||||
"dtd-content",
|
||||
f"<{element}> is empty. Either fill it or remove it.",
|
||||
)
|
||||
return (
|
||||
"dtd-content",
|
||||
f"<{element}> can't hold what's inside it here.",
|
||||
)
|
||||
|
||||
match = BAD_VALUE.match(message)
|
||||
if match:
|
||||
value, attribute, element = match.groups()
|
||||
if (element, attribute) == ("link", "rel"):
|
||||
return (
|
||||
"link-rel",
|
||||
f'The link with rel="{value}" isn\'t allowed. A head may hold '
|
||||
"the stylesheet link, an icon and a feed link.",
|
||||
)
|
||||
if (element, attribute) == ("body", "class"):
|
||||
return (
|
||||
"dtd-variant",
|
||||
f'"{value}" isn\'t a body variant. Use mews-warm, mews-cool, '
|
||||
"mews-green or mews-mono, or leave class out.",
|
||||
)
|
||||
allowed = ", ".join(_enumerated_values(element, attribute))
|
||||
return (
|
||||
"dtd-value",
|
||||
f'"{value}" isn\'t allowed for {attribute} on <{element}>. '
|
||||
+ (f"Use one of: {allowed}." if allowed else "Remove it."),
|
||||
)
|
||||
|
||||
match = MISSING_ATTRIBUTE.match(message)
|
||||
if match:
|
||||
element, attribute = match.groups()
|
||||
if element == "img" and attribute == "alt":
|
||||
return (
|
||||
"dtd-required-attribute",
|
||||
'Every image needs an alt attribute: a description, or alt="" '
|
||||
"when the image is decoration.",
|
||||
)
|
||||
return (
|
||||
"dtd-required-attribute",
|
||||
f"<{element}> needs a {attribute} attribute.",
|
||||
)
|
||||
|
||||
return ("dtd-other", f"This page doesn't match the Mews DTD ({message}).")
|
||||
|
||||
|
||||
# --- The rule checks -----------------------------------------------------
|
||||
|
||||
|
||||
def _prologue(report: Report, data: bytes) -> tuple[bool, bool]:
|
||||
"""Check the bytes before the root element.
|
||||
|
||||
Returns whether the page can be parsed at all and whether its entities may
|
||||
be expanded. This runs first on purpose: a page carrying its own entity
|
||||
declarations is refused before any parser sees them.
|
||||
"""
|
||||
head = data[:2048]
|
||||
if head.startswith(b"\xef\xbb\xbf"):
|
||||
report.add(
|
||||
"xml-declaration",
|
||||
"Remove the byte order mark at the start of the file. A Mews page "
|
||||
"starts with the XML declaration.",
|
||||
)
|
||||
head = head[3:]
|
||||
if not XML_DECLARATION.match(head):
|
||||
report.add(
|
||||
"xml-declaration",
|
||||
'Start the page with <?xml version="1.0" encoding="UTF-8"?>.',
|
||||
)
|
||||
|
||||
start = head.find(b"<!DOCTYPE")
|
||||
if start == -1:
|
||||
report.add(
|
||||
"doctype",
|
||||
"Add the XHTML Mobile 1.2 doctype after the XML declaration, as "
|
||||
"section 3.1 shows.",
|
||||
)
|
||||
return True, False
|
||||
|
||||
end = head.find(b">", start)
|
||||
if b"[" in head[start : end if end != -1 else len(head)]:
|
||||
report.add(
|
||||
"internal-subset",
|
||||
"Remove the extra declarations from the doctype. A Mews page uses "
|
||||
"the doctype from section 3.1 and nothing else.",
|
||||
)
|
||||
return False, False
|
||||
|
||||
if not DOCTYPE.match(head, start):
|
||||
report.add(
|
||||
"doctype",
|
||||
"Use the doctype from section 3.1 exactly, pointing at the XHTML "
|
||||
"Mobile 1.2 DTD.",
|
||||
)
|
||||
return True, False
|
||||
return True, True
|
||||
|
||||
|
||||
def _document_findings(report: Report, root: etree._Element) -> None:
|
||||
"""Check the root element and the head against sections 3.1 to 3.3."""
|
||||
if root.tag != f"{{{XHTML}}}html":
|
||||
report.add(
|
||||
"root-element",
|
||||
'The root element must be <html xmlns="http://www.w3.org/1999/xhtml">.',
|
||||
)
|
||||
return
|
||||
|
||||
xml_lang, lang = root.get(XML_LANG), root.get("lang")
|
||||
if not xml_lang or not lang or xml_lang != lang:
|
||||
report.add(
|
||||
"lang",
|
||||
"Name the page language with matching xml:lang and lang attributes "
|
||||
"on <html>, so browsers and clients both know it.",
|
||||
)
|
||||
report.language = xml_lang or lang or ""
|
||||
|
||||
head = root.find(f"{{{XHTML}}}head")
|
||||
if head is None:
|
||||
return
|
||||
|
||||
title = head.find(f"{{{XHTML}}}title")
|
||||
if title is not None and title.text:
|
||||
report.title = _clean(title.text)
|
||||
|
||||
markers = []
|
||||
for meta in head.iterfind(f"{{{XHTML}}}meta"):
|
||||
name = meta.get("name") or ""
|
||||
if name not in ALLOWED_META:
|
||||
report.add(
|
||||
"meta-name",
|
||||
f'The meta element named "{name}" isn\'t allowed. Section 3.3 '
|
||||
"lists the ones a head may hold.",
|
||||
)
|
||||
continue
|
||||
if name == MARKER:
|
||||
markers.append(meta.get("content") or "")
|
||||
elif name == "description":
|
||||
report.description = _clean(meta.get("content") or "")
|
||||
|
||||
if len(markers) > 1:
|
||||
report.add(
|
||||
"marker",
|
||||
'The head carries the <meta name="mews-profile" /> marker '
|
||||
f"{len(markers)} times. Keep one.",
|
||||
)
|
||||
elif not markers:
|
||||
report.add(
|
||||
"marker",
|
||||
'Add <meta name="mews-profile" content="0.1" /> to the head. '
|
||||
"Clients use it to tell a Mews page apart.",
|
||||
)
|
||||
elif markers[0] != VERSION:
|
||||
report.add(
|
||||
"marker-version",
|
||||
f'The marker says version "{markers[0]}". This validator checks '
|
||||
f"version {VERSION}.",
|
||||
)
|
||||
|
||||
viewport = any(
|
||||
meta.get("name") == "viewport" for meta in head.iterfind(f"{{{XHTML}}}meta")
|
||||
)
|
||||
if not viewport:
|
||||
report.add(
|
||||
"viewport",
|
||||
'Add <meta name="viewport" content="width=device-width" /> so phones '
|
||||
"show the page at a readable size.",
|
||||
)
|
||||
|
||||
|
||||
def _link_findings(
|
||||
report: Report, head: etree._Element, host: str | None
|
||||
) -> tuple[str | None, bool]:
|
||||
"""Check the head's links, returning the stylesheet href and if it's off-site."""
|
||||
stylesheets: list[str] = []
|
||||
for link in head.iterfind(f"{{{XHTML}}}link"):
|
||||
rel = (link.get("rel") or "").lower()
|
||||
href = link.get("href") or ""
|
||||
if rel not in ALLOWED_RELS:
|
||||
# Reported by the DTD check, which also names the three kinds.
|
||||
continue
|
||||
if rel == "stylesheet":
|
||||
stylesheets.append(href)
|
||||
elif rel == "alternate":
|
||||
kind = (link.get("type") or "").split(";")[0].strip()
|
||||
if kind != "application/atom+xml":
|
||||
report.add(
|
||||
"link-alternate-type",
|
||||
'A feed link needs type="application/atom+xml".',
|
||||
)
|
||||
elif rel == "icon" and host and not _is_same_site(host, href):
|
||||
report.add(
|
||||
"icon-offsite",
|
||||
"The icon is on another site. Host it on your own site, so "
|
||||
"reading a page tells no one else about it.",
|
||||
href,
|
||||
)
|
||||
|
||||
if len(stylesheets) > 1:
|
||||
report.add(
|
||||
"link-stylesheet-count",
|
||||
"Link the default stylesheet once. A Mews page has no other stylesheet.",
|
||||
)
|
||||
if not stylesheets:
|
||||
report.add(
|
||||
"stylesheet-missing",
|
||||
"Link your copy of mews-0.1.css, so readers without a Mews client "
|
||||
"still get good presentation.",
|
||||
)
|
||||
return None, False
|
||||
|
||||
href = stylesheets[0]
|
||||
if host and not _is_same_site(host, href):
|
||||
if _absolute(href).rstrip("/") == CANONICAL_STYLESHEET:
|
||||
report.add(
|
||||
"stylesheet-canonical",
|
||||
"Copy mews-0.1.css to your own site and link that copy, so no "
|
||||
"single server sees traffic across every Mews site.",
|
||||
href,
|
||||
)
|
||||
else:
|
||||
report.add(
|
||||
"stylesheet-offsite",
|
||||
"The stylesheet is on another site. A Mews page loads nothing "
|
||||
"from other sites. Copy mews-0.1.css to your own site.",
|
||||
href,
|
||||
)
|
||||
return href, True
|
||||
return href, False
|
||||
|
||||
|
||||
def _clean(text: str) -> str:
|
||||
"""Collapse whitespace and drop characters that could reshape a listing."""
|
||||
# Bidirectional overrides are printable but can make a title read as a
|
||||
# different domain, so they go too.
|
||||
overrides = frozenset(
|
||||
chr(code) for code in (*range(0x202A, 0x202F), *range(0x2066, 0x206A))
|
||||
)
|
||||
stripped = "".join(
|
||||
character
|
||||
for character in text
|
||||
if character.isprintable() and character not in overrides
|
||||
)
|
||||
return " ".join(stripped.split())
|
||||
|
||||
|
||||
def _is_same_site(host: str, href: str) -> bool:
|
||||
"""Say whether an href stays on the page's registered domain."""
|
||||
other = urlsplit(href).hostname
|
||||
return other is None or same_site(host, other)
|
||||
|
||||
|
||||
def _absolute(href: str) -> str:
|
||||
"""Treat an href with no scheme as https, for comparing with a known URL."""
|
||||
return href if "://" in href else "https://" + href.lstrip("/")
|
||||
|
||||
|
||||
def _transport_findings(report: Report, response: Fetched) -> None:
|
||||
"""Check the response headers against sections 7.1 to 7.4."""
|
||||
if "set-cookie" in response.headers:
|
||||
report.add(
|
||||
"cookie",
|
||||
"The server sets a cookie on this page. A Mews page sets cookies "
|
||||
"only for a form the reader submits.",
|
||||
)
|
||||
kind = (response.headers.get("content-type") or "").split(";")[0].strip().lower()
|
||||
if kind not in (
|
||||
"text/html",
|
||||
"application/xhtml+xml",
|
||||
"application/vnd.wap.xhtml+xml",
|
||||
):
|
||||
report.add(
|
||||
"content-type",
|
||||
f'The server sends this page as "{kind or "nothing"}". Send it as '
|
||||
"text/html, so a browser still shows a page with a small error.",
|
||||
)
|
||||
if not response.headers.get("last-modified") and not response.headers.get("etag"):
|
||||
report.add(
|
||||
"validators",
|
||||
"Send a Last-Modified or ETag header, so clients and the directory "
|
||||
"can check for changes cheaply.",
|
||||
)
|
||||
if urlsplit(response.url).scheme != "https" or response.scheme_downgraded:
|
||||
report.add(
|
||||
"https",
|
||||
"Serve the page over HTTPS. You may serve plain HTTP in parallel "
|
||||
"for old handsets.",
|
||||
)
|
||||
|
||||
|
||||
def _image_findings(
|
||||
report: Report, root: etree._Element, url: str | None, fetcher: Fetcher | None
|
||||
) -> int:
|
||||
"""Check every image, returning the bytes the fetched ones took."""
|
||||
host = urlsplit(url).hostname if url else None
|
||||
images = list(root.iter(f"{{{XHTML}}}img"))
|
||||
total = 0
|
||||
fetched = 0
|
||||
for image in images:
|
||||
src = (image.get("src") or "").strip()
|
||||
if src.lower().startswith("data:"):
|
||||
report.add(
|
||||
"image-data-uri",
|
||||
"This image is built into the page as a data: URI. Save it as a "
|
||||
"file on your own site and link it.",
|
||||
)
|
||||
continue
|
||||
if not src:
|
||||
continue
|
||||
target = urljoin(url, src) if url else src
|
||||
if host and not _is_same_site(host, target):
|
||||
report.add(
|
||||
"image-offsite",
|
||||
"This image is on another site. Host it on your own site, so "
|
||||
"reading a page tells no one else about it.",
|
||||
src,
|
||||
)
|
||||
continue
|
||||
if fetcher is None or url is None:
|
||||
continue
|
||||
if fetched >= MAX_IMAGES:
|
||||
continue
|
||||
fetched += 1
|
||||
total += _check_image(report, target, src, fetcher)
|
||||
|
||||
if fetcher is not None and len(images) > MAX_IMAGES:
|
||||
report.add(
|
||||
"images-not-checked",
|
||||
f"This page has {len(images)} images and the checker read the first "
|
||||
f"{MAX_IMAGES}. Check the rest yourself.",
|
||||
)
|
||||
return total
|
||||
|
||||
|
||||
def _check_image(report: Report, target: str, src: str, fetcher: Fetcher) -> int:
|
||||
"""Fetch one image and check its format, size and metadata."""
|
||||
try:
|
||||
response = fetcher.get(target, cap=IMAGE_CAP)
|
||||
except (UrlError, FetchError) as error:
|
||||
report.add("image-unreachable", f"This image didn't load: {error}", src)
|
||||
return 0
|
||||
if response.status != OK:
|
||||
report.add(
|
||||
"image-unreachable",
|
||||
f"This image didn't load (it returned {response.status}).",
|
||||
src,
|
||||
)
|
||||
return 0
|
||||
if "set-cookie" in response.headers:
|
||||
report.add("cookie", "The server sets a cookie on this image.", src)
|
||||
|
||||
body = response.body
|
||||
if response.truncated:
|
||||
report.add(
|
||||
"image-size",
|
||||
f"This image is over {IMAGE_CAP // 1024} KB. Keep images under "
|
||||
"50 KB, so a page loads quickly on a slow connection.",
|
||||
src,
|
||||
)
|
||||
elif len(body) > IMAGE_SHOULD:
|
||||
report.add(
|
||||
"image-size",
|
||||
f"This image is {len(body) // 1024} KB. Keep images under 50 KB, so "
|
||||
"a page loads quickly on a slow connection.",
|
||||
src,
|
||||
)
|
||||
|
||||
if not any(body.startswith(magic) for magic in IMAGE_MAGIC):
|
||||
report.add(
|
||||
"image-format",
|
||||
"This image isn't a GIF, JPEG or PNG. Use one of those, so every "
|
||||
"client and old handset can show it.",
|
||||
src,
|
||||
)
|
||||
elif _has_metadata(body):
|
||||
report.add(
|
||||
"image-metadata",
|
||||
"This image carries camera or location metadata. Strip it before "
|
||||
"publishing.",
|
||||
src,
|
||||
)
|
||||
return len(body)
|
||||
|
||||
|
||||
def _has_metadata(body: bytes) -> bool:
|
||||
"""Say whether an image carries an Exif block."""
|
||||
if body.startswith(b"\xff\xd8\xff"):
|
||||
return b"Exif\x00\x00" in body[:4096]
|
||||
if body.startswith(b"\x89PNG"):
|
||||
return b"eXIf" in body[:4096]
|
||||
return False
|
||||
|
||||
|
||||
def _stylesheet_findings(report: Report, href: str, url: str, fetcher: Fetcher) -> None:
|
||||
"""Fetch the linked stylesheet and compare it with the default one."""
|
||||
target = urljoin(url, href)
|
||||
host = urlsplit(url).hostname or ""
|
||||
try:
|
||||
response = fetcher.get(target, cap=CSS_CAP)
|
||||
except (UrlError, FetchError) as error:
|
||||
report.add(
|
||||
"stylesheet-unreachable", f"The stylesheet didn't load: {error}", href
|
||||
)
|
||||
return
|
||||
if response.status != OK:
|
||||
report.add(
|
||||
"stylesheet-unreachable",
|
||||
f"The stylesheet didn't load (it returned {response.status}).",
|
||||
href,
|
||||
)
|
||||
return
|
||||
if "set-cookie" in response.headers:
|
||||
report.add("cookie", "The server sets a cookie on the stylesheet.", href)
|
||||
|
||||
comparison = mews_css.compare(response.body.decode("utf-8", "replace"))
|
||||
if comparison.has_import:
|
||||
report.add(
|
||||
"stylesheet-import",
|
||||
"Your stylesheet copy uses @import. Remove it: a Mews page loads "
|
||||
"one stylesheet and nothing else.",
|
||||
href,
|
||||
)
|
||||
if not comparison.equal:
|
||||
report.add(
|
||||
"stylesheet-modified",
|
||||
"Your copy of mews-0.1.css has been changed, starting around line "
|
||||
f"{comparison.diff_line}. Copy it again unchanged. You may add "
|
||||
"@font-face rules that load fonts from your own site.",
|
||||
href,
|
||||
)
|
||||
for font in comparison.font_urls:
|
||||
if not _is_same_site(host, urljoin(target, font)):
|
||||
report.add(
|
||||
"font-offsite",
|
||||
"An @font-face rule loads a font from another site. Host the "
|
||||
"font file on your own site.",
|
||||
font,
|
||||
)
|
||||
|
||||
|
||||
# --- Entry points --------------------------------------------------------
|
||||
|
||||
|
||||
def validate_bytes(
|
||||
data: bytes,
|
||||
*,
|
||||
url: str | None = None,
|
||||
response: Fetched | None = None,
|
||||
fetcher: Fetcher | None = None,
|
||||
) -> Report:
|
||||
"""Validate one page.
|
||||
|
||||
url is the address the page was fetched from, which the same-site rules in
|
||||
4.2 and 7.4 need. Without a fetcher the checks that need the live page are
|
||||
skipped and said to be skipped.
|
||||
"""
|
||||
report = Report(url=url)
|
||||
report.size = len(data)
|
||||
|
||||
can_parse, expand_entities = _prologue(report, data)
|
||||
if not can_parse:
|
||||
return report
|
||||
|
||||
try:
|
||||
root = etree.fromstring(data, _parser(expand_entities=expand_entities))
|
||||
except etree.XMLSyntaxError as error:
|
||||
report.add(
|
||||
"well-formed",
|
||||
"This page isn't well-formed XML, so a Mews client can't read it "
|
||||
f"({error.msg}).",
|
||||
f"line {error.lineno}",
|
||||
)
|
||||
return report
|
||||
tree = etree.ElementTree(root)
|
||||
|
||||
_document_findings(report, root)
|
||||
_dtd_findings(report, tree)
|
||||
|
||||
host = urlsplit(url).hostname if url else None
|
||||
head = root.find(f"{{{XHTML}}}head")
|
||||
stylesheet, offsite = (None, False)
|
||||
if head is not None:
|
||||
stylesheet, offsite = _link_findings(report, head, host)
|
||||
|
||||
if response is not None:
|
||||
_transport_findings(report, response)
|
||||
|
||||
image_bytes = _image_findings(report, root, url, fetcher)
|
||||
|
||||
if stylesheet and not offsite and url and fetcher is not None:
|
||||
_stylesheet_findings(report, stylesheet, url, fetcher)
|
||||
|
||||
if (response is not None and response.truncated) or report.size > SIZE_MUST:
|
||||
report.add(
|
||||
"size-must",
|
||||
f"The page markup is over {SIZE_MUST // 1024} KB. Split it into "
|
||||
"several pages.",
|
||||
)
|
||||
elif report.size > SIZE_SHOULD:
|
||||
report.add(
|
||||
"size-should",
|
||||
f"The page markup is {report.size // 1024} KB. Keep it under "
|
||||
"64 KB, so it loads quickly on a slow connection.",
|
||||
)
|
||||
if fetcher is not None and report.size + image_bytes > SIZE_TOTAL:
|
||||
report.add(
|
||||
"size-total",
|
||||
"The page and its images come to more than 320 KB together. Use "
|
||||
"fewer or smaller images.",
|
||||
)
|
||||
|
||||
if fetcher is None:
|
||||
report.notes.append(
|
||||
"The checks that need the live page were skipped: response headers "
|
||||
"(7.1 to 7.4), image sizes and formats (4.2), and the stylesheet "
|
||||
"copy (5.2)."
|
||||
)
|
||||
return report
|
||||
|
||||
|
||||
def validate_file(path: str) -> Report:
|
||||
"""Validate a page on disk."""
|
||||
with open(path, "rb") as handle:
|
||||
return validate_bytes(handle.read())
|
||||
|
||||
|
||||
def validate_url(
|
||||
url: str, *, fetcher: Fetcher | None = None, allow_loopback: bool = False
|
||||
) -> Report:
|
||||
"""Fetch a page and validate it, along with its stylesheet and images."""
|
||||
owned = fetcher is None
|
||||
client = fetcher or Fetcher(allow_loopback=allow_loopback)
|
||||
try:
|
||||
response = client.get(normalise_url(url))
|
||||
if response.status != OK:
|
||||
report = Report(url=response.url)
|
||||
report.add(
|
||||
"well-formed",
|
||||
f"We couldn't read that page (it returned {response.status}). "
|
||||
"Check the address and try again.",
|
||||
)
|
||||
return report
|
||||
return validate_bytes(
|
||||
response.body, url=response.url, response=response, fetcher=client
|
||||
)
|
||||
finally:
|
||||
if owned:
|
||||
client.close()
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
"""Check pages named on the command line, returning 1 if any doesn't conform."""
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Check a page against Mews Profile 0.1 (see doc/SPEC.md)."
|
||||
)
|
||||
parser.add_argument("target", nargs="+", help="a page address or a file")
|
||||
parser.add_argument(
|
||||
"--local",
|
||||
action="store_true",
|
||||
help="also read addresses on this machine, for checking a site before "
|
||||
"it is published",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--quiet", action="store_true", help="print nothing for pages that conform"
|
||||
)
|
||||
args = parser.parse_args(argv)
|
||||
|
||||
worst = 0
|
||||
for target in args.target:
|
||||
if "://" in target or not _looks_like_path(target):
|
||||
try:
|
||||
report = validate_url(target, allow_loopback=args.local)
|
||||
except (UrlError, FetchError) as error:
|
||||
print(f"{target} — {error}", file=sys.stderr)
|
||||
worst = 1
|
||||
continue
|
||||
else:
|
||||
report = validate_file(target)
|
||||
|
||||
if report.conforms and args.quiet:
|
||||
continue
|
||||
worst = max(worst, 0 if report.conforms else 1)
|
||||
verdict = "conforms" if report.conforms else "doesn't conform"
|
||||
print(f"{target} — {verdict}")
|
||||
for finding in report.failures:
|
||||
print(f" {finding}")
|
||||
if report.warnings:
|
||||
print(" Worth fixing:")
|
||||
for finding in report.warnings:
|
||||
print(f" {finding}")
|
||||
for note in report.notes:
|
||||
print(f" Note: {note}")
|
||||
return worst
|
||||
|
||||
|
||||
def _looks_like_path(target: str) -> bool:
|
||||
"""Say whether a bare argument names a file rather than a site."""
|
||||
return Path(target).exists()
|
||||
Loading…
Add table
Add a link
Reference in a new issue