feat: pelican plugin

This commit is contained in:
randogoth 2026-10-11 08:09:18 +03:00
parent 970e0da9bc
commit 370fdfb067
33 changed files with 2274 additions and 0 deletions

View file

@ -0,0 +1 @@
from .mews import * # noqa: F403,PGH004,RUF100

View file

@ -0,0 +1,87 @@
<!-- Placeholder for the Mews subset definition. The real allowlist comes
from the Mews Profile DTD; replace this file when it is available.
Only element and attribute names are read; content models are
illustrative. -->
<!ENTITY % common
"id ID #IMPLIED
title CDATA #IMPLIED
xml:lang NMTOKEN #IMPLIED">
<!ELEMENT html (head, body) >
<!ATTLIST html
xmlns CDATA #FIXED "http://www.w3.org/1999/xhtml"
lang NMTOKEN #IMPLIED
dir (ltr | rtl) #IMPLIED
%common;
>
<!ELEMENT head (title, meta, link)* >
<!ELEMENT title (#PCDATA) >
<!ATTLIST title %common; >
<!ELEMENT meta EMPTY >
<!ATTLIST meta
name NMTOKEN #IMPLIED
content CDATA #REQUIRED
>
<!ELEMENT link EMPTY >
<!ATTLIST link
rel CDATA #IMPLIED
href CDATA #IMPLIED
type CDATA #IMPLIED
title CDATA #IMPLIED
%common;
>
<!ELEMENT body (#PCDATA | h1 | h2 | h3 | h4 | h5 | h6 | p | ul | blockquote | pre | hr | a | img)* >
<!ATTLIST body %common; >
<!ELEMENT h1 (#PCDATA) >
<!ATTLIST h1 %common; >
<!ELEMENT h2 (#PCDATA) >
<!ATTLIST h2 %common; >
<!ELEMENT h3 (#PCDATA) >
<!ATTLIST h3 %common; >
<!ELEMENT h4 (#PCDATA) >
<!ATTLIST h4 %common; >
<!ELEMENT h5 (#PCDATA) >
<!ATTLIST h5 %common; >
<!ELEMENT h6 (#PCDATA) >
<!ATTLIST h6 %common; >
<!ELEMENT p (#PCDATA | em | strong | code | a | img | br)* >
<!ATTLIST p %common; >
<!ELEMENT em (#PCDATA) >
<!ATTLIST em %common; >
<!ELEMENT strong (#PCDATA) >
<!ATTLIST strong %common; >
<!ELEMENT code (#PCDATA) >
<!ATTLIST code %common; >
<!ELEMENT a (#PCDATA) >
<!ATTLIST a
href CDATA #IMPLIED
title CDATA #IMPLIED
rel CDATA #IMPLIED
%common;
>
<!ELEMENT img EMPTY >
<!ATTLIST img
src CDATA #IMPLIED
alt CDATA #IMPLIED
%common;
>
<!ELEMENT br EMPTY >
<!ELEMENT ul (li)+ >
<!ATTLIST ul %common; >
<!ELEMENT li (#PCDATA | p | ul)* >
<!ATTLIST li %common; >
<!ELEMENT blockquote (p)* >
<!ATTLIST blockquote %common; >
<!ELEMENT pre (#PCDATA) >
<!ATTLIST pre %common; >
<!ELEMENT hr EMPTY >

View file

@ -0,0 +1,127 @@
"""Make Pelican output Mews Profile pages.
Every .xhtml file Pelican writes is brought inside the subset the
bundled Mews DTD defines: disallowed markup is stripped (or, with
MEWS_STRICT, fails the build), the XHTML namespace is ensured, and the
page is re-serialized as well-formed UTF-8 XML. The plugin also copies the
Mews stylesheet into the output tree and hands themes a MEWS_CSS global
and the mews/head.html partial.
Settings:
MEWS_STRICT Fail the build on disallowed content instead of stripping
it. Defaults to False. Removals are always logged.
MEWS_CSS_DIR Directory below the output root for the stylesheet. Defaults
to "css"; themes get the path as the MEWS_CSS global.
MEWS_OUTPUT_PATH Put the finished pages in this folder instead of
OUTPUT_PATH, so a normal blog can keep its own output tree.
Unset by default. The stylesheet is copied there too, and the
.xhtml file Pelican left in OUTPUT_PATH is removed.
"""
from __future__ import annotations
from importlib.resources import files
import logging
import os
from pathlib import Path
import shutil
from lxml import etree
from pelican import signals
from .subset import (
MewsSubsetError,
element_name,
enforce_subset,
ensure_xhtml_root,
parse_allowlist,
parse_page,
serialize,
)
__all__ = ("register",)
LOGGER = logging.getLogger(__name__)
CSS_NAME = "mews-0.1.css"
DEFAULT_CSS_DIR = "css"
XHTML_SUFFIX = ".xhtml"
ALLOWLIST = parse_allowlist(
files(__package__).joinpath("dtd", "mews.dtd").read_text(encoding="utf-8")
)
# content_written carries no settings, so the handler reads them from here;
# refreshed by the initialized signal, which fires before every build.
_settings: dict = {}
def _on_initialized(pelican) -> None:
_settings.clear()
_settings.update(pelican.settings)
overrides = pelican.settings.setdefault("THEME_TEMPLATES_OVERRIDES", [])
templates = str(files(__package__).joinpath("templates"))
if templates not in overrides:
overrides.append(templates)
def _on_generator_init(generator) -> None:
css_dir = generator.settings.get("MEWS_CSS_DIR", DEFAULT_CSS_DIR)
generator.env.globals["MEWS_CSS"] = f"{css_dir}/{CSS_NAME}"
def _on_content_written(path, **_) -> None:
file = Path(path)
if file.suffix != XHTML_SUFFIX:
return
try:
tree, doctype = parse_page(file.read_bytes())
except etree.XMLSyntaxError as error:
message = f"{file} is not well-formed XML: {error}"
raise MewsSubsetError(message) from error
if element_name(tree.getroot().tag) != "html":
LOGGER.warning("%s: root element is not <html>; leaving it unchanged", file)
return
tree = ensure_xhtml_root(tree)
strict = bool(_settings.get("MEWS_STRICT", False))
enforce_subset(tree, ALLOWLIST, strict=strict, source=str(file), log=LOGGER.warning)
data = serialize(tree, doctype)
parallel = _settings.get("MEWS_OUTPUT_PATH")
if not parallel:
file.write_bytes(data)
return
target = Path(parallel) / os.path.relpath(file, _settings["OUTPUT_PATH"])
target.parent.mkdir(parents=True, exist_ok=True)
target.write_bytes(data)
file.unlink()
_prune_empty_dirs(file, Path(_settings["OUTPUT_PATH"]).resolve())
LOGGER.info("moved %s to %s", file, target)
def _prune_empty_dirs(path: Path, root: Path) -> None:
"""Remove directories the move left empty, up to the output root."""
empty = path.parent.resolve()
while empty not in (root, empty.parent) and not any(empty.iterdir()):
empty.rmdir()
empty = empty.parent
def _on_finalized(pelican) -> None:
css_dir = pelican.settings.get("MEWS_CSS_DIR", DEFAULT_CSS_DIR)
root = pelican.settings.get("MEWS_OUTPUT_PATH") or pelican.settings["OUTPUT_PATH"]
target = Path(root) / css_dir / CSS_NAME
target.parent.mkdir(parents=True, exist_ok=True)
source = files(__package__).joinpath("static", CSS_NAME)
with source.open("rb") as src, target.open("wb") as dst:
shutil.copyfileobj(src, dst)
LOGGER.info("wrote %s", target)
def register() -> None:
"""Connect the plugin to Pelican's signals."""
signals.initialized.connect(_on_initialized)
signals.generator_init.connect(_on_generator_init)
signals.content_written.connect(_on_content_written)
signals.finalized.connect(_on_finalized)

View file

@ -0,0 +1,106 @@
/* Mews Profile default stylesheet 0.1 */
/* Base: light theme */
body {
background-color: #faf8f3;
color: #1f1f1f;
font-family: "Atkinson Hyperlegible", Verdana, Tahoma, system-ui, sans-serif;
font-size: 106%;
line-height: 1.5;
margin: 0 auto;
padding: 1em;
max-width: 40em;
}
h1, h2, h3, h4, h5, h6 {
font-weight: bold;
line-height: 1.25;
margin: 1.6em 0 0.5em 0;
}
h1 { font-size: 1.6em; margin-top: 0.5em; }
h2 { font-size: 1.3em; }
h3 { font-size: 1.1em; }
h4, h5, h6 { font-size: 1em; }
p, ul, ol, dl, blockquote, pre, table, address {
margin: 0 0 1em 0;
}
ul, ol { padding-left: 1.5em; }
li { margin-bottom: 0.25em; }
dt { font-weight: bold; }
dd { margin: 0 0 0.5em 1.5em; }
a { color: #1a4f9c; text-decoration: underline; }
a:visited { color: #5b2a86; }
a:focus, a:active { outline: 2px solid #1a4f9c; }
blockquote {
margin-left: 0;
padding-left: 1em;
border-left: 3px solid #c9c4b8;
}
pre, code, kbd, samp {
font-family: "Source Code Pro", Consolas, Menlo, "DejaVu Sans Mono", monospace;
font-size: 0.9em;
}
pre {
background-color: #efece4;
padding: 0.75em;
overflow: auto;
}
hr {
border: 0;
border-top: 1px solid #c9c4b8;
margin: 2em 0;
}
table { border-collapse: collapse; }
th, td {
border: 1px solid #c9c4b8;
padding: 0.3em 0.6em;
text-align: left;
vertical-align: top;
}
caption { font-weight: bold; text-align: left; }
img {
display: block;
max-width: 100%;
height: auto;
border: 0;
}
input, select, textarea { font-size: 1em; font-family: inherit; }
/* Body variants: light */
body.mews-warm { background-color: #fbf4e8; color: #2a2420; }
body.mews-warm a { color: #8a3d14; }
body.mews-cool { background-color: #f3f6fa; color: #1c2430; }
body.mews-cool a { color: #1a4f9c; }
body.mews-green { background-color: #f2f6ef; color: #1f2a1f; }
body.mews-green a { color: #24502a; }
body.mews-mono { background-color: #e8ebe1; color: #1e2219; }
body.mews-mono a { color: #1e2219; }
/* Dark theme: follows the system setting */
@media (prefers-color-scheme: dark) {
body { background-color: #18191b; color: #e3e1dc; }
a { color: #8fb8f5; }
a:visited { color: #c9a3f2; }
a:focus, a:active { outline-color: #8fb8f5; }
blockquote, hr, th, td { border-color: #45464a; }
pre { background-color: #222326; }
body.mews-warm { background-color: #1d1916; color: #e8dfd3; }
body.mews-warm a { color: #f0b48a; }
body.mews-cool { background-color: #161a20; color: #dde4ee; }
body.mews-cool a { color: #8fb8f5; }
body.mews-green { background-color: #161b16; color: #dbe6d8; }
body.mews-green a { color: #9fd39a; }
body.mews-mono { background-color: #1b1d18; color: #c8d0b8; }
body.mews-mono a { color: #c8d0b8; }
}
html { color-scheme: light dark; }

View file

@ -0,0 +1,179 @@
"""The Mews subset: the allowlist and the transform enforcing it.
The allowlist is read from the DTD, the only source of what is allowed.
Element names come from <!ELEMENT> declarations, attributes from
<!ATTLIST> groups, with parameter entities expanded so groups such as
%Common.attrib; contribute the attributes they stand for.
"""
from __future__ import annotations
from collections.abc import Callable
from io import BytesIO
import re
from typing import TypeAlias
from lxml import etree
XHTML_NS = "http://www.w3.org/1999/xhtml"
# lxml represents the xml:lang attribute with this Clark-notation key.
XML_NS = "{http://www.w3.org/XML/1998/namespace}"
Tree: TypeAlias = etree._ElementTree
Element: TypeAlias = etree._Element
Allowlist: TypeAlias = dict[str, set[str]]
# An attribute name followed by a type token, e.g. "href CDATA #IMPLIED".
_ATTRIBUTE = re.compile(
r"([\w.:-]+)\s+(?:CDATA|NMTOKENS?|ID|IDREFS?|ENTIT(?:IES|Y)|NOTATION|\()"
)
_ENTITY = re.compile(
r"<!ENTITY\s+%\s+([\w.-]+)\s+(?:PUBLIC|SYSTEM)?\s*\"(.*?)\"\s*>", re.DOTALL
)
_ATTLIST = re.compile(r"<!ATTLIST\s+([\w.:-]+)(.*?)>", re.DOTALL)
_ELEMENT = re.compile(r"<!ELEMENT\s+([\w.:-]+)")
_REFERENCE = re.compile(r"%([\w.-]+);")
class MewsSubsetError(RuntimeError):
"""Content outside the Mews subset, in strict mode."""
def parse_allowlist(dtd: str) -> Allowlist:
"""Map element name to its allowed attributes, read from DTD markup."""
entities = dict(_ENTITY.findall(dtd))
def expand(text: str) -> str:
for _ in range(len(entities) + 1):
expanded = _REFERENCE.sub(
lambda match: entities.get(match.group(1), match.group(0)), text
)
if expanded == text:
break
text = expanded
return text
allowlist: Allowlist = {}
for name, body in _ATTLIST.findall(dtd):
allowlist.setdefault(name, set()).update(_ATTRIBUTE.findall(expand(body)))
for name in _ELEMENT.findall(dtd):
allowlist.setdefault(name, set())
return allowlist
def element_name(tag: str) -> str:
"""Local name of a tag, with any namespace stripped."""
return tag.rpartition("}")[2] if tag.startswith("{") else tag
def attribute_name(key: str) -> str:
"""Name of an attribute key, mapping the xml namespace to the xml: prefix."""
if key.startswith(XML_NS):
return "xml:" + key[len(XML_NS) :]
return key
def parse_page(data: bytes) -> tuple[Tree, str | None]:
"""Parse one page, returning its tree and original doctype."""
parser = etree.XMLParser(resolve_entities=False, load_dtd=False)
tree = etree.parse(BytesIO(data), parser)
return tree, tree.docinfo.doctype or None
def ensure_xhtml_root(tree: Tree) -> Tree:
"""Give <html> the XHTML default namespace if it is missing.
Children keep their own tags; a default namespace declared on the
root element covers them.
"""
root = tree.getroot()
if root.nsmap.get(None) == XHTML_NS:
return tree
html = etree.Element(f"{{{XHTML_NS}}}html")
html.attrib.update(root.attrib)
html.text, html.tail = root.text, root.tail
for child in root:
html.append(child)
return etree.ElementTree(html)
def serialize(tree: Tree, doctype: str | None) -> bytes:
"""Serialize as UTF-8 XML; empty elements come out self-closed."""
return etree.tostring(tree, xml_declaration=True, encoding="UTF-8", doctype=doctype)
def enforce_subset(
tree: Tree,
allowlist: Allowlist,
*,
strict: bool,
source: str,
log: Callable[[str], None],
) -> None:
"""Keep only what the allowlist permits.
Disallowed elements are unwrapped: their children move up, so no text
is lost. Comments and processing instructions are dropped. Every
removal is reported through log; with strict set, any violation raises
instead of changing the page.
"""
violations: list[str] = []
for element in list(tree.getroot().iter()):
if not isinstance(element.tag, str): # comment or processing instruction
if not strict:
_remove(element)
continue
name = element_name(element.tag)
if name not in allowlist:
violations.append(f"{source}: <{name}> is not in the Mews subset")
if not strict:
_unwrap(element)
continue
allowed = allowlist[name]
for key in list(element.attrib):
if attribute_name(key) not in allowed:
violations.append(
f"{source}: {name} {attribute_name(key)} is not in the Mews subset"
)
if not strict:
del element.attrib[key]
if not violations:
return
for message in violations:
log(message)
if strict:
message = "content outside the Mews subset; refusing to write it:\n "
raise MewsSubsetError(message + "\n ".join(violations))
def _unwrap(element: Element) -> None:
"""Replace an element with its children, keeping the text flow."""
parent = element.getparent()
index = parent.index(element)
children = list(element)
if children:
children[0].text = (element.text or "") + (children[0].text or "")
children[-1].tail = (children[-1].tail or "") + (element.tail or "")
for child in children:
parent.insert(index, child)
index += 1
parent.remove(element)
return
previous = element.getprevious()
text = (element.text or "") + (element.tail or "")
if previous is not None:
previous.tail = (previous.tail or "") + text
else:
parent.text = (parent.text or "") + text
parent.remove(element)
def _remove(element: Element) -> None:
"""Drop a comment or processing instruction, keeping the text flow."""
parent = element.getparent()
previous = element.getprevious()
if previous is not None:
previous.tail = (previous.tail or "") + (element.tail or "")
else:
parent.text = (parent.text or "") + (element.tail or "")
parent.remove(element)

View file

@ -0,0 +1,3 @@
<meta name="mews-profile" content="0.1" />
<meta name="viewport" content="width=device-width" />
<link rel="stylesheet" type="text/css" href="{{ MEWS_CSS }}" />