feat: pelican plugin
This commit is contained in:
parent
970e0da9bc
commit
370fdfb067
33 changed files with 2274 additions and 0 deletions
1
pelican-mews/pelican/plugins/mews/__init__.py
Normal file
1
pelican-mews/pelican/plugins/mews/__init__.py
Normal file
|
|
@ -0,0 +1 @@
|
|||
from .mews import * # noqa: F403,PGH004,RUF100
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
87
pelican-mews/pelican/plugins/mews/dtd/mews.dtd
Normal file
87
pelican-mews/pelican/plugins/mews/dtd/mews.dtd
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
<!-- Placeholder for the Mews subset definition. The real allowlist comes
|
||||
from the Mews Profile DTD; replace this file when it is available.
|
||||
Only element and attribute names are read; content models are
|
||||
illustrative. -->
|
||||
|
||||
<!ENTITY % common
|
||||
"id ID #IMPLIED
|
||||
title CDATA #IMPLIED
|
||||
xml:lang NMTOKEN #IMPLIED">
|
||||
|
||||
<!ELEMENT html (head, body) >
|
||||
<!ATTLIST html
|
||||
xmlns CDATA #FIXED "http://www.w3.org/1999/xhtml"
|
||||
lang NMTOKEN #IMPLIED
|
||||
dir (ltr | rtl) #IMPLIED
|
||||
%common;
|
||||
>
|
||||
|
||||
<!ELEMENT head (title, meta, link)* >
|
||||
<!ELEMENT title (#PCDATA) >
|
||||
<!ATTLIST title %common; >
|
||||
|
||||
<!ELEMENT meta EMPTY >
|
||||
<!ATTLIST meta
|
||||
name NMTOKEN #IMPLIED
|
||||
content CDATA #REQUIRED
|
||||
>
|
||||
|
||||
<!ELEMENT link EMPTY >
|
||||
<!ATTLIST link
|
||||
rel CDATA #IMPLIED
|
||||
href CDATA #IMPLIED
|
||||
type CDATA #IMPLIED
|
||||
title CDATA #IMPLIED
|
||||
%common;
|
||||
>
|
||||
|
||||
<!ELEMENT body (#PCDATA | h1 | h2 | h3 | h4 | h5 | h6 | p | ul | blockquote | pre | hr | a | img)* >
|
||||
<!ATTLIST body %common; >
|
||||
|
||||
<!ELEMENT h1 (#PCDATA) >
|
||||
<!ATTLIST h1 %common; >
|
||||
<!ELEMENT h2 (#PCDATA) >
|
||||
<!ATTLIST h2 %common; >
|
||||
<!ELEMENT h3 (#PCDATA) >
|
||||
<!ATTLIST h3 %common; >
|
||||
<!ELEMENT h4 (#PCDATA) >
|
||||
<!ATTLIST h4 %common; >
|
||||
<!ELEMENT h5 (#PCDATA) >
|
||||
<!ATTLIST h5 %common; >
|
||||
<!ELEMENT h6 (#PCDATA) >
|
||||
<!ATTLIST h6 %common; >
|
||||
|
||||
<!ELEMENT p (#PCDATA | em | strong | code | a | img | br)* >
|
||||
<!ATTLIST p %common; >
|
||||
|
||||
<!ELEMENT em (#PCDATA) >
|
||||
<!ATTLIST em %common; >
|
||||
<!ELEMENT strong (#PCDATA) >
|
||||
<!ATTLIST strong %common; >
|
||||
<!ELEMENT code (#PCDATA) >
|
||||
<!ATTLIST code %common; >
|
||||
<!ELEMENT a (#PCDATA) >
|
||||
<!ATTLIST a
|
||||
href CDATA #IMPLIED
|
||||
title CDATA #IMPLIED
|
||||
rel CDATA #IMPLIED
|
||||
%common;
|
||||
>
|
||||
<!ELEMENT img EMPTY >
|
||||
<!ATTLIST img
|
||||
src CDATA #IMPLIED
|
||||
alt CDATA #IMPLIED
|
||||
%common;
|
||||
>
|
||||
<!ELEMENT br EMPTY >
|
||||
|
||||
<!ELEMENT ul (li)+ >
|
||||
<!ATTLIST ul %common; >
|
||||
<!ELEMENT li (#PCDATA | p | ul)* >
|
||||
<!ATTLIST li %common; >
|
||||
|
||||
<!ELEMENT blockquote (p)* >
|
||||
<!ATTLIST blockquote %common; >
|
||||
<!ELEMENT pre (#PCDATA) >
|
||||
<!ATTLIST pre %common; >
|
||||
<!ELEMENT hr EMPTY >
|
||||
127
pelican-mews/pelican/plugins/mews/mews.py
Normal file
127
pelican-mews/pelican/plugins/mews/mews.py
Normal file
|
|
@ -0,0 +1,127 @@
|
|||
"""Make Pelican output Mews Profile pages.
|
||||
|
||||
Every .xhtml file Pelican writes is brought inside the subset the
|
||||
bundled Mews DTD defines: disallowed markup is stripped (or, with
|
||||
MEWS_STRICT, fails the build), the XHTML namespace is ensured, and the
|
||||
page is re-serialized as well-formed UTF-8 XML. The plugin also copies the
|
||||
Mews stylesheet into the output tree and hands themes a MEWS_CSS global
|
||||
and the mews/head.html partial.
|
||||
|
||||
Settings:
|
||||
|
||||
MEWS_STRICT Fail the build on disallowed content instead of stripping
|
||||
it. Defaults to False. Removals are always logged.
|
||||
MEWS_CSS_DIR Directory below the output root for the stylesheet. Defaults
|
||||
to "css"; themes get the path as the MEWS_CSS global.
|
||||
MEWS_OUTPUT_PATH Put the finished pages in this folder instead of
|
||||
OUTPUT_PATH, so a normal blog can keep its own output tree.
|
||||
Unset by default. The stylesheet is copied there too, and the
|
||||
.xhtml file Pelican left in OUTPUT_PATH is removed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from importlib.resources import files
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
|
||||
from lxml import etree
|
||||
|
||||
from pelican import signals
|
||||
|
||||
from .subset import (
|
||||
MewsSubsetError,
|
||||
element_name,
|
||||
enforce_subset,
|
||||
ensure_xhtml_root,
|
||||
parse_allowlist,
|
||||
parse_page,
|
||||
serialize,
|
||||
)
|
||||
|
||||
__all__ = ("register",)
|
||||
|
||||
LOGGER = logging.getLogger(__name__)
|
||||
|
||||
CSS_NAME = "mews-0.1.css"
|
||||
DEFAULT_CSS_DIR = "css"
|
||||
XHTML_SUFFIX = ".xhtml"
|
||||
|
||||
ALLOWLIST = parse_allowlist(
|
||||
files(__package__).joinpath("dtd", "mews.dtd").read_text(encoding="utf-8")
|
||||
)
|
||||
|
||||
# content_written carries no settings, so the handler reads them from here;
|
||||
# refreshed by the initialized signal, which fires before every build.
|
||||
_settings: dict = {}
|
||||
|
||||
|
||||
def _on_initialized(pelican) -> None:
|
||||
_settings.clear()
|
||||
_settings.update(pelican.settings)
|
||||
overrides = pelican.settings.setdefault("THEME_TEMPLATES_OVERRIDES", [])
|
||||
templates = str(files(__package__).joinpath("templates"))
|
||||
if templates not in overrides:
|
||||
overrides.append(templates)
|
||||
|
||||
|
||||
def _on_generator_init(generator) -> None:
|
||||
css_dir = generator.settings.get("MEWS_CSS_DIR", DEFAULT_CSS_DIR)
|
||||
generator.env.globals["MEWS_CSS"] = f"{css_dir}/{CSS_NAME}"
|
||||
|
||||
|
||||
def _on_content_written(path, **_) -> None:
|
||||
file = Path(path)
|
||||
if file.suffix != XHTML_SUFFIX:
|
||||
return
|
||||
try:
|
||||
tree, doctype = parse_page(file.read_bytes())
|
||||
except etree.XMLSyntaxError as error:
|
||||
message = f"{file} is not well-formed XML: {error}"
|
||||
raise MewsSubsetError(message) from error
|
||||
if element_name(tree.getroot().tag) != "html":
|
||||
LOGGER.warning("%s: root element is not <html>; leaving it unchanged", file)
|
||||
return
|
||||
tree = ensure_xhtml_root(tree)
|
||||
strict = bool(_settings.get("MEWS_STRICT", False))
|
||||
enforce_subset(tree, ALLOWLIST, strict=strict, source=str(file), log=LOGGER.warning)
|
||||
data = serialize(tree, doctype)
|
||||
parallel = _settings.get("MEWS_OUTPUT_PATH")
|
||||
if not parallel:
|
||||
file.write_bytes(data)
|
||||
return
|
||||
target = Path(parallel) / os.path.relpath(file, _settings["OUTPUT_PATH"])
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
target.write_bytes(data)
|
||||
file.unlink()
|
||||
_prune_empty_dirs(file, Path(_settings["OUTPUT_PATH"]).resolve())
|
||||
LOGGER.info("moved %s to %s", file, target)
|
||||
|
||||
|
||||
def _prune_empty_dirs(path: Path, root: Path) -> None:
|
||||
"""Remove directories the move left empty, up to the output root."""
|
||||
empty = path.parent.resolve()
|
||||
while empty not in (root, empty.parent) and not any(empty.iterdir()):
|
||||
empty.rmdir()
|
||||
empty = empty.parent
|
||||
|
||||
|
||||
def _on_finalized(pelican) -> None:
|
||||
css_dir = pelican.settings.get("MEWS_CSS_DIR", DEFAULT_CSS_DIR)
|
||||
root = pelican.settings.get("MEWS_OUTPUT_PATH") or pelican.settings["OUTPUT_PATH"]
|
||||
target = Path(root) / css_dir / CSS_NAME
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
source = files(__package__).joinpath("static", CSS_NAME)
|
||||
with source.open("rb") as src, target.open("wb") as dst:
|
||||
shutil.copyfileobj(src, dst)
|
||||
LOGGER.info("wrote %s", target)
|
||||
|
||||
|
||||
def register() -> None:
|
||||
"""Connect the plugin to Pelican's signals."""
|
||||
signals.initialized.connect(_on_initialized)
|
||||
signals.generator_init.connect(_on_generator_init)
|
||||
signals.content_written.connect(_on_content_written)
|
||||
signals.finalized.connect(_on_finalized)
|
||||
0
pelican-mews/pelican/plugins/mews/py.typed
Normal file
0
pelican-mews/pelican/plugins/mews/py.typed
Normal file
106
pelican-mews/pelican/plugins/mews/static/mews-0.1.css
Normal file
106
pelican-mews/pelican/plugins/mews/static/mews-0.1.css
Normal file
|
|
@ -0,0 +1,106 @@
|
|||
/* Mews Profile default stylesheet 0.1 */
|
||||
|
||||
/* Base: light theme */
|
||||
body {
|
||||
background-color: #faf8f3;
|
||||
color: #1f1f1f;
|
||||
font-family: "Atkinson Hyperlegible", Verdana, Tahoma, system-ui, sans-serif;
|
||||
font-size: 106%;
|
||||
line-height: 1.5;
|
||||
margin: 0 auto;
|
||||
padding: 1em;
|
||||
max-width: 40em;
|
||||
}
|
||||
|
||||
h1, h2, h3, h4, h5, h6 {
|
||||
font-weight: bold;
|
||||
line-height: 1.25;
|
||||
margin: 1.6em 0 0.5em 0;
|
||||
}
|
||||
h1 { font-size: 1.6em; margin-top: 0.5em; }
|
||||
h2 { font-size: 1.3em; }
|
||||
h3 { font-size: 1.1em; }
|
||||
h4, h5, h6 { font-size: 1em; }
|
||||
|
||||
p, ul, ol, dl, blockquote, pre, table, address {
|
||||
margin: 0 0 1em 0;
|
||||
}
|
||||
ul, ol { padding-left: 1.5em; }
|
||||
li { margin-bottom: 0.25em; }
|
||||
dt { font-weight: bold; }
|
||||
dd { margin: 0 0 0.5em 1.5em; }
|
||||
|
||||
a { color: #1a4f9c; text-decoration: underline; }
|
||||
a:visited { color: #5b2a86; }
|
||||
a:focus, a:active { outline: 2px solid #1a4f9c; }
|
||||
|
||||
blockquote {
|
||||
margin-left: 0;
|
||||
padding-left: 1em;
|
||||
border-left: 3px solid #c9c4b8;
|
||||
}
|
||||
|
||||
pre, code, kbd, samp {
|
||||
font-family: "Source Code Pro", Consolas, Menlo, "DejaVu Sans Mono", monospace;
|
||||
font-size: 0.9em;
|
||||
}
|
||||
pre {
|
||||
background-color: #efece4;
|
||||
padding: 0.75em;
|
||||
overflow: auto;
|
||||
}
|
||||
|
||||
hr {
|
||||
border: 0;
|
||||
border-top: 1px solid #c9c4b8;
|
||||
margin: 2em 0;
|
||||
}
|
||||
|
||||
table { border-collapse: collapse; }
|
||||
th, td {
|
||||
border: 1px solid #c9c4b8;
|
||||
padding: 0.3em 0.6em;
|
||||
text-align: left;
|
||||
vertical-align: top;
|
||||
}
|
||||
caption { font-weight: bold; text-align: left; }
|
||||
|
||||
img {
|
||||
display: block;
|
||||
max-width: 100%;
|
||||
height: auto;
|
||||
border: 0;
|
||||
}
|
||||
|
||||
input, select, textarea { font-size: 1em; font-family: inherit; }
|
||||
|
||||
/* Body variants: light */
|
||||
body.mews-warm { background-color: #fbf4e8; color: #2a2420; }
|
||||
body.mews-warm a { color: #8a3d14; }
|
||||
body.mews-cool { background-color: #f3f6fa; color: #1c2430; }
|
||||
body.mews-cool a { color: #1a4f9c; }
|
||||
body.mews-green { background-color: #f2f6ef; color: #1f2a1f; }
|
||||
body.mews-green a { color: #24502a; }
|
||||
body.mews-mono { background-color: #e8ebe1; color: #1e2219; }
|
||||
body.mews-mono a { color: #1e2219; }
|
||||
|
||||
/* Dark theme: follows the system setting */
|
||||
@media (prefers-color-scheme: dark) {
|
||||
body { background-color: #18191b; color: #e3e1dc; }
|
||||
a { color: #8fb8f5; }
|
||||
a:visited { color: #c9a3f2; }
|
||||
a:focus, a:active { outline-color: #8fb8f5; }
|
||||
blockquote, hr, th, td { border-color: #45464a; }
|
||||
pre { background-color: #222326; }
|
||||
|
||||
body.mews-warm { background-color: #1d1916; color: #e8dfd3; }
|
||||
body.mews-warm a { color: #f0b48a; }
|
||||
body.mews-cool { background-color: #161a20; color: #dde4ee; }
|
||||
body.mews-cool a { color: #8fb8f5; }
|
||||
body.mews-green { background-color: #161b16; color: #dbe6d8; }
|
||||
body.mews-green a { color: #9fd39a; }
|
||||
body.mews-mono { background-color: #1b1d18; color: #c8d0b8; }
|
||||
body.mews-mono a { color: #c8d0b8; }
|
||||
}
|
||||
|
||||
html { color-scheme: light dark; }
|
||||
179
pelican-mews/pelican/plugins/mews/subset.py
Normal file
179
pelican-mews/pelican/plugins/mews/subset.py
Normal file
|
|
@ -0,0 +1,179 @@
|
|||
"""The Mews subset: the allowlist and the transform enforcing it.
|
||||
|
||||
The allowlist is read from the DTD, the only source of what is allowed.
|
||||
Element names come from <!ELEMENT> declarations, attributes from
|
||||
<!ATTLIST> groups, with parameter entities expanded so groups such as
|
||||
%Common.attrib; contribute the attributes they stand for.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Callable
|
||||
from io import BytesIO
|
||||
import re
|
||||
from typing import TypeAlias
|
||||
|
||||
from lxml import etree
|
||||
|
||||
XHTML_NS = "http://www.w3.org/1999/xhtml"
|
||||
# lxml represents the xml:lang attribute with this Clark-notation key.
|
||||
XML_NS = "{http://www.w3.org/XML/1998/namespace}"
|
||||
|
||||
Tree: TypeAlias = etree._ElementTree
|
||||
Element: TypeAlias = etree._Element
|
||||
Allowlist: TypeAlias = dict[str, set[str]]
|
||||
|
||||
# An attribute name followed by a type token, e.g. "href CDATA #IMPLIED".
|
||||
_ATTRIBUTE = re.compile(
|
||||
r"([\w.:-]+)\s+(?:CDATA|NMTOKENS?|ID|IDREFS?|ENTIT(?:IES|Y)|NOTATION|\()"
|
||||
)
|
||||
_ENTITY = re.compile(
|
||||
r"<!ENTITY\s+%\s+([\w.-]+)\s+(?:PUBLIC|SYSTEM)?\s*\"(.*?)\"\s*>", re.DOTALL
|
||||
)
|
||||
_ATTLIST = re.compile(r"<!ATTLIST\s+([\w.:-]+)(.*?)>", re.DOTALL)
|
||||
_ELEMENT = re.compile(r"<!ELEMENT\s+([\w.:-]+)")
|
||||
_REFERENCE = re.compile(r"%([\w.-]+);")
|
||||
|
||||
|
||||
class MewsSubsetError(RuntimeError):
|
||||
"""Content outside the Mews subset, in strict mode."""
|
||||
|
||||
|
||||
def parse_allowlist(dtd: str) -> Allowlist:
|
||||
"""Map element name to its allowed attributes, read from DTD markup."""
|
||||
entities = dict(_ENTITY.findall(dtd))
|
||||
|
||||
def expand(text: str) -> str:
|
||||
for _ in range(len(entities) + 1):
|
||||
expanded = _REFERENCE.sub(
|
||||
lambda match: entities.get(match.group(1), match.group(0)), text
|
||||
)
|
||||
if expanded == text:
|
||||
break
|
||||
text = expanded
|
||||
return text
|
||||
|
||||
allowlist: Allowlist = {}
|
||||
for name, body in _ATTLIST.findall(dtd):
|
||||
allowlist.setdefault(name, set()).update(_ATTRIBUTE.findall(expand(body)))
|
||||
for name in _ELEMENT.findall(dtd):
|
||||
allowlist.setdefault(name, set())
|
||||
return allowlist
|
||||
|
||||
|
||||
def element_name(tag: str) -> str:
|
||||
"""Local name of a tag, with any namespace stripped."""
|
||||
return tag.rpartition("}")[2] if tag.startswith("{") else tag
|
||||
|
||||
|
||||
def attribute_name(key: str) -> str:
|
||||
"""Name of an attribute key, mapping the xml namespace to the xml: prefix."""
|
||||
if key.startswith(XML_NS):
|
||||
return "xml:" + key[len(XML_NS) :]
|
||||
return key
|
||||
|
||||
|
||||
def parse_page(data: bytes) -> tuple[Tree, str | None]:
|
||||
"""Parse one page, returning its tree and original doctype."""
|
||||
parser = etree.XMLParser(resolve_entities=False, load_dtd=False)
|
||||
tree = etree.parse(BytesIO(data), parser)
|
||||
return tree, tree.docinfo.doctype or None
|
||||
|
||||
|
||||
def ensure_xhtml_root(tree: Tree) -> Tree:
|
||||
"""Give <html> the XHTML default namespace if it is missing.
|
||||
|
||||
Children keep their own tags; a default namespace declared on the
|
||||
root element covers them.
|
||||
"""
|
||||
root = tree.getroot()
|
||||
if root.nsmap.get(None) == XHTML_NS:
|
||||
return tree
|
||||
html = etree.Element(f"{{{XHTML_NS}}}html")
|
||||
html.attrib.update(root.attrib)
|
||||
html.text, html.tail = root.text, root.tail
|
||||
for child in root:
|
||||
html.append(child)
|
||||
return etree.ElementTree(html)
|
||||
|
||||
|
||||
def serialize(tree: Tree, doctype: str | None) -> bytes:
|
||||
"""Serialize as UTF-8 XML; empty elements come out self-closed."""
|
||||
return etree.tostring(tree, xml_declaration=True, encoding="UTF-8", doctype=doctype)
|
||||
|
||||
|
||||
def enforce_subset(
|
||||
tree: Tree,
|
||||
allowlist: Allowlist,
|
||||
*,
|
||||
strict: bool,
|
||||
source: str,
|
||||
log: Callable[[str], None],
|
||||
) -> None:
|
||||
"""Keep only what the allowlist permits.
|
||||
|
||||
Disallowed elements are unwrapped: their children move up, so no text
|
||||
is lost. Comments and processing instructions are dropped. Every
|
||||
removal is reported through log; with strict set, any violation raises
|
||||
instead of changing the page.
|
||||
"""
|
||||
violations: list[str] = []
|
||||
for element in list(tree.getroot().iter()):
|
||||
if not isinstance(element.tag, str): # comment or processing instruction
|
||||
if not strict:
|
||||
_remove(element)
|
||||
continue
|
||||
name = element_name(element.tag)
|
||||
if name not in allowlist:
|
||||
violations.append(f"{source}: <{name}> is not in the Mews subset")
|
||||
if not strict:
|
||||
_unwrap(element)
|
||||
continue
|
||||
allowed = allowlist[name]
|
||||
for key in list(element.attrib):
|
||||
if attribute_name(key) not in allowed:
|
||||
violations.append(
|
||||
f"{source}: {name} {attribute_name(key)} is not in the Mews subset"
|
||||
)
|
||||
if not strict:
|
||||
del element.attrib[key]
|
||||
if not violations:
|
||||
return
|
||||
for message in violations:
|
||||
log(message)
|
||||
if strict:
|
||||
message = "content outside the Mews subset; refusing to write it:\n "
|
||||
raise MewsSubsetError(message + "\n ".join(violations))
|
||||
|
||||
|
||||
def _unwrap(element: Element) -> None:
|
||||
"""Replace an element with its children, keeping the text flow."""
|
||||
parent = element.getparent()
|
||||
index = parent.index(element)
|
||||
children = list(element)
|
||||
if children:
|
||||
children[0].text = (element.text or "") + (children[0].text or "")
|
||||
children[-1].tail = (children[-1].tail or "") + (element.tail or "")
|
||||
for child in children:
|
||||
parent.insert(index, child)
|
||||
index += 1
|
||||
parent.remove(element)
|
||||
return
|
||||
previous = element.getprevious()
|
||||
text = (element.text or "") + (element.tail or "")
|
||||
if previous is not None:
|
||||
previous.tail = (previous.tail or "") + text
|
||||
else:
|
||||
parent.text = (parent.text or "") + text
|
||||
parent.remove(element)
|
||||
|
||||
|
||||
def _remove(element: Element) -> None:
|
||||
"""Drop a comment or processing instruction, keeping the text flow."""
|
||||
parent = element.getparent()
|
||||
previous = element.getprevious()
|
||||
if previous is not None:
|
||||
previous.tail = (previous.tail or "") + (element.tail or "")
|
||||
else:
|
||||
parent.text = (parent.text or "") + (element.tail or "")
|
||||
parent.remove(element)
|
||||
|
|
@ -0,0 +1,3 @@
|
|||
<meta name="mews-profile" content="0.1" />
|
||||
<meta name="viewport" content="width=device-width" />
|
||||
<link rel="stylesheet" type="text/css" href="{{ MEWS_CSS }}" />
|
||||
Loading…
Add table
Add a link
Reference in a new issue