feat: add xhtmlmp renderer
This commit is contained in:
parent
3ed0b5807c
commit
3af44fb0ac
14 changed files with 524 additions and 7 deletions
|
|
@ -19,6 +19,7 @@ from .plugins import (
|
|||
register_parser,
|
||||
)
|
||||
from .renderers import wml as _wml # noqa: F401 # registers the WML renderer plugin
|
||||
from .renderers import xhtmlmp as _xhtmlmp # noqa: F401 # registers the XHTML-MP renderer plugin
|
||||
from .renderers.wml.renderer import DeckOutput
|
||||
|
||||
|
||||
|
|
@ -211,8 +212,21 @@ def _cli_overrides(args: argparse.Namespace) -> Dict[str, Any]:
|
|||
return {dest: getattr(args, dest, None) for dest in _CONFIG_DESTS}
|
||||
|
||||
|
||||
def write_output(output: DeckOutput, path: Optional[Path]) -> None:
|
||||
"""Write the deck. WML is XML served over HTTP, so lines end with LF."""
|
||||
def write_output(output: "DeckOutput | str", path: Optional[Path]) -> None:
|
||||
"""Write the renderer's output. WML/XHTML-MP are XML served over HTTP,
|
||||
so lines end with LF.
|
||||
|
||||
A plain string (as XHTML-MP produces -- it has no deck/card topology
|
||||
to report) is written as-is; a DeckOutput may fan out to several files
|
||||
in --deck-per-card mode.
|
||||
"""
|
||||
if isinstance(output, str):
|
||||
if path is None:
|
||||
sys.stdout.write(output)
|
||||
return
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(output, encoding="utf-8")
|
||||
return
|
||||
multi = len(output.documents) > 1 or any(doc.name for doc in output.documents)
|
||||
if multi:
|
||||
if path is None:
|
||||
|
|
|
|||
|
|
@ -1,5 +1,6 @@
|
|||
"""Bundled renderer implementations."""
|
||||
|
||||
from .wml import WMLRenderer
|
||||
from .xhtmlmp import XhtmlMpRenderer
|
||||
|
||||
__all__ = ["WMLRenderer"]
|
||||
__all__ = ["WMLRenderer", "XhtmlMpRenderer"]
|
||||
|
|
|
|||
24
src/wapdown/renderers/xhtmlmp/__init__.py
Normal file
24
src/wapdown/renderers/xhtmlmp/__init__.py
Normal file
|
|
@ -0,0 +1,24 @@
|
|||
"""XHTML-MP renderer, registered under the name `xhtmlmp`."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
from ...plugins import register_renderer
|
||||
from .inline import escape_xhtml, process_inline
|
||||
from .renderer import XHTML_PROLOG, XhtmlMpRenderer
|
||||
|
||||
|
||||
def _xhtmlmp_renderer_factory(**options: Any) -> XhtmlMpRenderer:
|
||||
return XhtmlMpRenderer(**options)
|
||||
|
||||
|
||||
try:
|
||||
register_renderer("xhtmlmp", _xhtmlmp_renderer_factory)
|
||||
except ValueError: # pragma: no cover - already registered on re-import
|
||||
pass
|
||||
|
||||
__all__ = [
|
||||
"XHTML_PROLOG",
|
||||
"XhtmlMpRenderer",
|
||||
"escape_xhtml",
|
||||
"process_inline",
|
||||
]
|
||||
95
src/wapdown/renderers/xhtmlmp/inline.py
Normal file
95
src/wapdown/renderers/xhtmlmp/inline.py
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
"""Inline Markdown -> XHTML-MP markup mapping, plus XML escaping.
|
||||
|
||||
Mirrors wml/inline.py's stash-then-substitute structure and reuses its
|
||||
Markdown-parsing regexes (a URL like `/a_b_c` must survive intact, which is
|
||||
what the stash step protects against), but the two output grammars differ
|
||||
enough that the escaping and link/code handling are not shared:
|
||||
|
||||
- XHTML-MP has no '$' variable-substitution quirk.
|
||||
- <a> may contain inline markup, unlike WML's `(#PCDATA | br | img)*`, so
|
||||
a link label is run through the same emphasis pass as ordinary text
|
||||
instead of having its markers stripped.
|
||||
- <code> exists, so a code span becomes an element instead of plain text.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import List
|
||||
|
||||
from ..wml.inline import (
|
||||
BOLD_RE,
|
||||
CODE_STASH_RE,
|
||||
IMAGE_RE,
|
||||
ITALIC_RE,
|
||||
LINK_RE,
|
||||
STRIKETHROUGH_RE,
|
||||
UNDERLINE_EM_RE,
|
||||
UNDERLINE_STRONG_RE,
|
||||
)
|
||||
|
||||
|
||||
_STASH = "\u0000{kind}{index}\u0000"
|
||||
_CODE = "CODE"
|
||||
_TAG = "TAG"
|
||||
|
||||
|
||||
def escape_xhtml(text: str) -> str:
|
||||
"""XML-escape text for XHTML-MP element content or a double-quoted
|
||||
attribute value. Applied once, up front, to the whole raw string before
|
||||
any inline markup is turned into tags, matching wml/inline.py's
|
||||
ordering, so link/image URLs captured out of the already-escaped text
|
||||
are correctly escaped for attribute use too.
|
||||
"""
|
||||
text = text.replace("&", "&")
|
||||
return text.replace("<", "<").replace(">", ">").replace('"', """)
|
||||
|
||||
|
||||
def process_inline(text: str) -> str:
|
||||
"""Convert one run of inline Markdown into XHTML-MP markup.
|
||||
|
||||
Links and images are resolved *before* the emphasis patterns run and
|
||||
stashed behind sentinels, for the same reason as in wml/inline.py: a
|
||||
path like `/a_b_c` or `/a*b*c` would otherwise be shredded into <em>
|
||||
tags by the emphasis regexes before LINK_RE ever saw it.
|
||||
"""
|
||||
escaped = escape_xhtml(text)
|
||||
|
||||
code_segments: List[str] = []
|
||||
tag_segments: List[str] = []
|
||||
|
||||
def stash_code(match):
|
||||
code_segments.append(match.group(0)[1:-1])
|
||||
return _STASH.format(kind=_CODE, index=len(code_segments) - 1)
|
||||
|
||||
def stash(markup: str) -> str:
|
||||
tag_segments.append(markup)
|
||||
return _STASH.format(kind=_TAG, index=len(tag_segments) - 1)
|
||||
|
||||
def stash_image(match):
|
||||
alt = match.group(1).strip()
|
||||
src = match.group(2).strip()
|
||||
return stash(f'<img src="{src}" alt="{alt}"/>')
|
||||
|
||||
def stash_link(match):
|
||||
label = match.group(1).strip()
|
||||
url = match.group(2).strip()
|
||||
return stash(f'<a href="{url}">{_apply_emphasis(label)}</a>')
|
||||
|
||||
escaped = CODE_STASH_RE.sub(stash_code, escaped)
|
||||
escaped = IMAGE_RE.sub(stash_image, escaped) # before LINK_RE: `` is not a link
|
||||
escaped = LINK_RE.sub(stash_link, escaped)
|
||||
escaped = _apply_emphasis(escaped)
|
||||
|
||||
for index, markup in enumerate(tag_segments):
|
||||
escaped = escaped.replace(_STASH.format(kind=_TAG, index=index), markup)
|
||||
for index, code in enumerate(code_segments):
|
||||
escaped = escaped.replace(_STASH.format(kind=_CODE, index=index), f"<code>{code}</code>")
|
||||
return escaped
|
||||
|
||||
|
||||
def _apply_emphasis(text: str) -> str:
|
||||
text = STRIKETHROUGH_RE.sub(lambda m: m.group(1), text) # no XHTML-MP equivalent
|
||||
text = BOLD_RE.sub(lambda m: f"<strong>{m.group(1)}</strong>", text)
|
||||
text = UNDERLINE_STRONG_RE.sub(lambda m: f"<strong>{m.group(1)}</strong>", text)
|
||||
text = ITALIC_RE.sub(lambda m: f"<em>{m.group(1)}</em>", text)
|
||||
text = UNDERLINE_EM_RE.sub(lambda m: f"<em>{m.group(1)}</em>", text)
|
||||
return text
|
||||
157
src/wapdown/renderers/xhtmlmp/renderer.py
Normal file
157
src/wapdown/renderers/xhtmlmp/renderer.py
Normal file
|
|
@ -0,0 +1,157 @@
|
|||
"""The XHTML-MP renderer."""
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any, List, Optional
|
||||
|
||||
from ...config import DeckConfig
|
||||
from ...models import BlockEvent, BlockKind, StyleUpdateEvent
|
||||
from .inline import escape_xhtml, process_inline
|
||||
|
||||
|
||||
XHTML_PROLOG = (
|
||||
'<?xml version="1.0" encoding="UTF-8"?>\n'
|
||||
'<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN" '
|
||||
'"http://www.wapforum.org/DTD/xhtml-mobile10.dtd">\n'
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class _ListFrame:
|
||||
"""One open <ul>/<ol>, mid-render.
|
||||
|
||||
`items` holds complete `<li>...</li>` strings. A deeper list is spliced
|
||||
into the last item *before* its closing tag when it closes, because
|
||||
XHTML nests a sub-list inside its parent <li>, not after it -- unlike
|
||||
WML, which has no list element to nest at all.
|
||||
"""
|
||||
|
||||
depth: int
|
||||
ordered: bool
|
||||
items: List[str] = field(default_factory=list)
|
||||
|
||||
|
||||
class XhtmlMpRenderer:
|
||||
"""Render Markdown events as an XHTML-MP page for WAP 2.0 browsers and,
|
||||
served as text/html, modern browsers too.
|
||||
|
||||
Unlike WMLRenderer, there is no deck-size budget or card topology to
|
||||
compute, so output streams incrementally per event rather than being
|
||||
buffered and resolved in `finalize()`.
|
||||
"""
|
||||
|
||||
def __init__(self, config: Optional[DeckConfig] = None, **options: Any) -> None:
|
||||
self.config = config or DeckConfig.resolve(overrides=options)
|
||||
self._body: List[str] = []
|
||||
self._list_stack: List[_ListFrame] = []
|
||||
|
||||
# -- Renderer protocol ---------------------------------------------------
|
||||
|
||||
def handle_event(self, event: "BlockEvent | StyleUpdateEvent") -> None:
|
||||
if isinstance(event, StyleUpdateEvent):
|
||||
# XHTML-MP has no per-block style stack to apply this to, by
|
||||
# design; ignored like other structural mismatches.
|
||||
return
|
||||
handler = {
|
||||
BlockKind.PARAGRAPH: self._render_paragraph,
|
||||
BlockKind.HEADING: self._render_heading,
|
||||
BlockKind.LIST_ITEM: self._render_list_item,
|
||||
BlockKind.CODE_BLOCK: self._render_code_block,
|
||||
BlockKind.BLOCKQUOTE: self._render_blockquote,
|
||||
BlockKind.TABLE: self._render_table,
|
||||
BlockKind.HORIZONTAL_RULE: self._render_horizontal_rule,
|
||||
BlockKind.CARD_BREAK: self._close_lists,
|
||||
# BLANK_LINE: a blank line between list items is loose Markdown
|
||||
# formatting, not a break in the list, so it leaves any open
|
||||
# list open -- matching the WML renderer's own convention of
|
||||
# not resetting list state on a blank line.
|
||||
}.get(event.kind)
|
||||
if handler is not None:
|
||||
handler(event.payload)
|
||||
|
||||
def finalize(self) -> str:
|
||||
self._close_lists()
|
||||
title = escape_xhtml(self.config.title) if self.config.title else ""
|
||||
head = f"<head><title>{title}</title></head>"
|
||||
body = "".join(self._body)
|
||||
return (
|
||||
f"{XHTML_PROLOG}"
|
||||
'<html xmlns="http://www.w3.org/1999/xhtml">\n'
|
||||
f"{head}\n"
|
||||
f"<body>{body}</body>\n"
|
||||
"</html>\n"
|
||||
)
|
||||
|
||||
# -- List nesting ---------------------------------------------------------
|
||||
|
||||
def _render_list_item(self, payload) -> None:
|
||||
indent = payload.indent.replace("\t", " ")
|
||||
depth = max(0, len(indent) // 2)
|
||||
text = self._inline(payload.text)
|
||||
self._open_list(depth, payload.ordered)
|
||||
self._list_stack[-1].items.append(f"<li>{text}</li>")
|
||||
|
||||
def _open_list(self, depth: int, ordered: bool) -> None:
|
||||
while self._list_stack and (
|
||||
self._list_stack[-1].depth > depth
|
||||
or (self._list_stack[-1].depth == depth and self._list_stack[-1].ordered != ordered)
|
||||
):
|
||||
self._close_one_list()
|
||||
if not self._list_stack or self._list_stack[-1].depth < depth:
|
||||
self._list_stack.append(_ListFrame(depth, ordered))
|
||||
|
||||
def _close_one_list(self) -> None:
|
||||
frame = self._list_stack.pop()
|
||||
tag = "ol" if frame.ordered else "ul"
|
||||
wrapped = f"<{tag}>{''.join(frame.items)}</{tag}>"
|
||||
if self._list_stack:
|
||||
parent_items = self._list_stack[-1].items
|
||||
parent_items[-1] = parent_items[-1][: -len("</li>")] + wrapped + "</li>"
|
||||
else:
|
||||
self._body.append(wrapped)
|
||||
|
||||
def _close_lists(self, *_: object) -> None:
|
||||
while self._list_stack:
|
||||
self._close_one_list()
|
||||
|
||||
# -- Block markup mapping -------------------------------------------------
|
||||
|
||||
def _render_paragraph(self, payload) -> None:
|
||||
self._close_lists()
|
||||
self._body.append(f"<p>{self._inline(payload.text)}</p>")
|
||||
|
||||
def _render_heading(self, payload) -> None:
|
||||
self._close_lists()
|
||||
level = max(1, min(6, payload.level))
|
||||
text = self._inline(payload.text)
|
||||
self._body.append(f"<h{level}>{text}</h{level}>")
|
||||
|
||||
def _render_code_block(self, payload) -> None:
|
||||
self._close_lists()
|
||||
lines = "\n".join(escape_xhtml(line.rstrip("\n")) for line in payload.lines)
|
||||
self._body.append(f"<pre>{lines}</pre>")
|
||||
|
||||
def _render_blockquote(self, payload) -> None:
|
||||
self._close_lists()
|
||||
self._body.append(f"<blockquote><p>{self._inline(payload.text)}</p></blockquote>")
|
||||
|
||||
def _render_table(self, payload) -> None:
|
||||
self._close_lists()
|
||||
rows = payload.rows
|
||||
if not rows:
|
||||
return
|
||||
columns = max(len(row) for row in rows)
|
||||
tr_parts: List[str] = []
|
||||
for row_index, row in enumerate(rows):
|
||||
cells = list(row) + [""] * (columns - len(row))
|
||||
cell_tag = "th" if row_index == 0 else "td"
|
||||
tds = "".join(f"<{cell_tag}>{self._inline(cell)}</{cell_tag}>" for cell in cells)
|
||||
tr_parts.append(f"<tr>{tds}</tr>")
|
||||
self._body.append(f"<table>{''.join(tr_parts)}</table>")
|
||||
|
||||
def _render_horizontal_rule(self, _payload) -> None:
|
||||
self._close_lists()
|
||||
self._body.append("<hr/>")
|
||||
|
||||
def _inline(self, text: str) -> str:
|
||||
return process_inline(text)
|
||||
|
|
@ -25,11 +25,11 @@ def render_source(text: str, **options: Any) -> str:
|
|||
return convert(body, config=config, base_path=EXAMPLES).text
|
||||
|
||||
|
||||
def render_file(path: Path, **options: Any) -> DeckOutput:
|
||||
def render_file(path: Path, *, renderer_name: str = "wml", **options: Any):
|
||||
lines = read_lines(path)
|
||||
frontmatter, body = strip_frontmatter(lines)
|
||||
config = DeckConfig.resolve(frontmatter, options)
|
||||
return convert(body, config=config, base_path=path.parent)
|
||||
return convert(body, config=config, base_path=path.parent, renderer_name=renderer_name)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
|
|
@ -37,6 +37,19 @@ def render():
|
|||
return render_source
|
||||
|
||||
|
||||
def render_xhtmlmp_source(text: str, **options: Any) -> str:
|
||||
"""Like `render_source`, but through the XHTML-MP renderer."""
|
||||
lines = text.splitlines(keepends=True)
|
||||
frontmatter, body = strip_frontmatter(lines)
|
||||
config = DeckConfig.resolve(frontmatter, options)
|
||||
return convert(body, config=config, base_path=EXAMPLES, renderer_name="xhtmlmp")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def render_xhtmlmp():
|
||||
return render_xhtmlmp_source
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def render_output():
|
||||
"""Like `render`, but exposes every output file (deck-per-card mode)."""
|
||||
|
|
|
|||
6
tests/golden/kitchen-sink-xhtmlmp.xhtml
Normal file
6
tests/golden/kitchen-sink-xhtmlmp.xhtml
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN" "http://www.wapforum.org/DTD/xhtml-mobile10.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml">
|
||||
<head><title>Kitchen Sink</title></head>
|
||||
<body><h1>Kitchen Sink</h1><p>A deck exercising every block and inline construct wapdown understands.</p><h2>Links and emphasis</h2><p>Tricky URL: <a href="https://example.com/docs/v1_2_3/notes.html">release notes</a> and a starred path <a href="https://ci.example.com/job/a*b*c/log">build log</a>.</p><p><strong>Bold</strong>, <em>italic</em>, <code>inline code</code>, struck, and a literal $5 fee.</p><h2>A table</h2><table><tr><th>Trail</th><th>Status</th><th>Fee</th></tr><tr><td>North Loop</td><td>open</td><td>$2</td></tr><tr><td>Summit Spur</td><td>icy</td><td>$5</td></tr></table><h2>Other blocks</h2><blockquote><p>A quoted warning.</p></blockquote><ol><li>First</li><li>Second</li></ol><hr/><pre>indented code $HOME</pre></body>
|
||||
</html>
|
||||
6
tests/golden/trail-manual-xhtmlmp.xhtml
Normal file
6
tests/golden/trail-manual-xhtmlmp.xhtml
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN" "http://www.wapforum.org/DTD/xhtml-mobile10.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml">
|
||||
<head><title></title></head>
|
||||
<body><p>Live updates for the ridge trail network. Reception is spotty past the tree line, check before you go.</p><p>Cold and clear. Wind: <strong>15 mph</strong> gusting from the <em>northwest</em>. Permit fee: $5.</p><ul><li>North Loop: open</li><li>South Loop: closed</li><li>Summit Spur: open, ice above 2000m</li></ul><p>Ranger station: <a href="tel:+15555550123">call dispatch</a></p><p><img src="https://example.com/map.png" alt="Trail map"/></p></body>
|
||||
</html>
|
||||
6
tests/golden/trail-rules-xhtmlmp.xhtml
Normal file
6
tests/golden/trail-rules-xhtmlmp.xhtml
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN" "http://www.wapforum.org/DTD/xhtml-mobile10.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml">
|
||||
<head><title></title></head>
|
||||
<body><p>Live updates for the ridge trail network. Reception is spotty past the tree line, check before you go.</p><hr/><h2>Weather</h2><p>Cold and clear. Wind: <strong>15 mph</strong> gusting from the <em>northwest</em>. Permit fee: $5.</p><hr/><h2>Trail Status</h2><ul><li>North Loop: open</li><li>South Loop: closed</li><li>Summit Spur: open, ice above 2000m</li></ul><hr/><h2>Contact</h2><p>Ranger station: <a href="tel:+15555550123">call dispatch</a></p><hr/></body>
|
||||
</html>
|
||||
6
tests/golden/trail-xhtmlmp.xhtml
Normal file
6
tests/golden/trail-xhtmlmp.xhtml
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN" "http://www.wapforum.org/DTD/xhtml-mobile10.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml">
|
||||
<head><title></title></head>
|
||||
<body><h1>Trail Conditions</h1><p>Live updates for the ridge trail network. Reception is spotty past the tree line, check before you go.</p><h2>Weather</h2><p>Cold and clear. Wind: <strong>15 mph</strong> gusting from the <em>northwest</em>. Permit fee: $5.</p><h2>Trail Status</h2><ul><li>North Loop: open</li><li>South Loop: closed</li><li>Summit Spur: open, ice above 2000m</li></ul><h2>Contact</h2><p>Ranger station: <a href="tel:+15555550123">call dispatch</a></p><p><img src="https://example.com/map.png" alt="Trail map"/></p></body>
|
||||
</html>
|
||||
|
|
@ -30,3 +30,14 @@ CASES: Dict[str, Tuple[str, Dict[str, Any]]] = {
|
|||
MULTI_CASES: Dict[str, Tuple[str, Dict[str, Any]]] = {
|
||||
"trail-decks": ("trail-manual.md", {"deck_per_card": True, "home_label": "Home"}),
|
||||
}
|
||||
|
||||
# name -> (example filename, renderer options), rendered through the
|
||||
# XHTML-MP renderer. A separate table from CASES because XHTML-MP output is
|
||||
# a plain string, not a DeckOutput -- there is no deck/card topology to
|
||||
# vary, so only options that affect ordinary block/inline rendering apply.
|
||||
XHTMLMP_CASES: Dict[str, Tuple[str, Dict[str, Any]]] = {
|
||||
"trail-xhtmlmp": ("trail.md", {}),
|
||||
"trail-manual-xhtmlmp": ("trail-manual.md", {}),
|
||||
"trail-rules-xhtmlmp": ("trail-rules.md", {}),
|
||||
"kitchen-sink-xhtmlmp": ("kitchen-sink.md", {}),
|
||||
}
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ from pathlib import Path
|
|||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
|
||||
from conftest import EXAMPLES, GOLDEN, render_file # noqa: E402
|
||||
from golden_cases import CASES, MULTI_CASES # noqa: E402
|
||||
from golden_cases import CASES, MULTI_CASES, XHTMLMP_CASES # noqa: E402
|
||||
|
||||
|
||||
def main() -> int:
|
||||
|
|
@ -37,6 +37,11 @@ def main() -> int:
|
|||
document.text, encoding="utf-8"
|
||||
)
|
||||
print(f"wrote {name}/ ({len(list(directory.iterdir()))} files)")
|
||||
|
||||
for name, (source, options) in XHTMLMP_CASES.items():
|
||||
output = render_file(EXAMPLES / source, renderer_name="xhtmlmp", **options)
|
||||
(GOLDEN / f"{name}.xhtml").write_text(output, encoding="utf-8")
|
||||
print(f"wrote {name}.xhtml")
|
||||
return 0
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ from xml.etree import ElementTree
|
|||
import pytest
|
||||
|
||||
from conftest import EXAMPLES, GOLDEN, render_file
|
||||
from golden_cases import CASES, MULTI_CASES
|
||||
from golden_cases import CASES, MULTI_CASES, XHTMLMP_CASES
|
||||
|
||||
|
||||
REGENERATE_HINT = (
|
||||
|
|
@ -49,6 +49,20 @@ def test_golden_files_are_well_formed(path):
|
|||
ElementTree.fromstring(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("name", sorted(XHTMLMP_CASES))
|
||||
def test_matches_golden_xhtmlmp(name):
|
||||
source, options = XHTMLMP_CASES[name]
|
||||
golden = GOLDEN / f"{name}.xhtml"
|
||||
assert golden.exists(), f"missing golden for '{name}'; {REGENERATE_HINT}"
|
||||
actual = render_file(EXAMPLES / source, renderer_name="xhtmlmp", **options)
|
||||
assert actual == golden.read_text(encoding="utf-8"), REGENERATE_HINT
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", sorted(GOLDEN.glob("**/*.xhtml")), ids=lambda p: p.name)
|
||||
def test_xhtmlmp_golden_files_are_well_formed(path):
|
||||
ElementTree.fromstring(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def test_every_example_has_a_case():
|
||||
"""A new examples/*.md without a golden case would go unverified."""
|
||||
covered = {source for source, _ in list(CASES.values()) + list(MULTI_CASES.values())}
|
||||
|
|
|
|||
159
tests/test_xhtmlmp.py
Normal file
159
tests/test_xhtmlmp.py
Normal file
|
|
@ -0,0 +1,159 @@
|
|||
"""XHTML-MP emission: prolog, head, block mapping, and well-formedness."""
|
||||
from __future__ import annotations
|
||||
|
||||
from xml.etree import ElementTree
|
||||
|
||||
import pytest
|
||||
|
||||
from wapdown.renderers.xhtmlmp.renderer import XHTML_PROLOG
|
||||
|
||||
|
||||
TWO_SECTIONS = "Intro.\n\n{.card One}\nFirst.\n\n{.card Two}\nSecond.\n"
|
||||
|
||||
|
||||
def parse(markup: str) -> ElementTree.Element:
|
||||
"""Parse the page, proving it is well-formed XML.
|
||||
|
||||
ElementTree honours the DOCTYPE declaration without trying to fetch the
|
||||
external DTD, so this never depends on a network round trip to
|
||||
wapforum.org.
|
||||
"""
|
||||
return ElementTree.fromstring(markup)
|
||||
|
||||
|
||||
class TestDocumentShape:
|
||||
def test_prolog_and_doctype(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("Body.\n")
|
||||
assert markup.startswith(XHTML_PROLOG)
|
||||
assert 'PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN"' in markup
|
||||
|
||||
def test_root_element_is_html(self, render_xhtmlmp):
|
||||
# ElementTree namespace-qualifies the tag because of the xhtml xmlns.
|
||||
assert parse(render_xhtmlmp("Body.\n")).tag == "{http://www.w3.org/1999/xhtml}html"
|
||||
|
||||
def test_output_is_well_formed(self, render_xhtmlmp):
|
||||
parse(render_xhtmlmp(TWO_SECTIONS))
|
||||
|
||||
def test_no_carriage_returns(self, render_xhtmlmp):
|
||||
assert "\r" not in render_xhtmlmp(TWO_SECTIONS)
|
||||
|
||||
def test_card_breaks_do_not_split_the_page(self, render_xhtmlmp):
|
||||
# Cards are a WAP 1.x screen-budget concern; XHTML-MP renders one
|
||||
# continuous page and a {.card} marker draws nothing of its own.
|
||||
markup = render_xhtmlmp(TWO_SECTIONS)
|
||||
assert markup.count("<html") == 1
|
||||
assert "First." in markup and "Second." in markup
|
||||
|
||||
def test_title_is_escaped_into_head(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("Body.\n", title="Fish & Chips")
|
||||
assert "<title>Fish & Chips</title>" in markup
|
||||
|
||||
|
||||
class TestBlockMapping:
|
||||
@pytest.mark.parametrize(
|
||||
("source", "expected"),
|
||||
[
|
||||
("Just text.\n", "<p>Just text.</p>"),
|
||||
("> quoted\n", "<blockquote><p>quoted</p></blockquote>"),
|
||||
("```\ncode\n```\n", "<pre>code</pre>"),
|
||||
("---\n", "<hr/>"),
|
||||
],
|
||||
)
|
||||
def test_blocks(self, render_xhtmlmp, source, expected):
|
||||
assert expected in render_xhtmlmp(source)
|
||||
|
||||
@pytest.mark.parametrize("level", [1, 2, 3, 4, 5, 6])
|
||||
def test_headings_use_real_heading_elements(self, render_xhtmlmp, level):
|
||||
markup = render_xhtmlmp(f"{'#' * level} Title\n")
|
||||
assert f"<h{level}>Title</h{level}>" in markup
|
||||
assert "<big>" not in markup # no FIGlet/WML-style banner substitute
|
||||
|
||||
def test_heading_level_beyond_six_clamps(self, render_xhtmlmp):
|
||||
# Markdown itself caps at h6, but a manually-crafted payload
|
||||
# shouldn't be able to emit an invalid element name.
|
||||
markup = render_xhtmlmp("###### Deep\n")
|
||||
assert "<h6>Deep</h6>" in markup
|
||||
|
||||
def test_unordered_list(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("- a\n- b\n")
|
||||
assert "<ul><li>a</li><li>b</li></ul>" in markup
|
||||
|
||||
def test_ordered_list(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("1. a\n2. b\n")
|
||||
assert "<ol><li>a</li><li>b</li></ol>" in markup
|
||||
|
||||
def test_nested_list_is_spliced_into_parent_item(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("- a\n - nested\n- b\n")
|
||||
# The sub-list must sit *inside* its parent <li>, not after it, or
|
||||
# the result is not a valid nested list.
|
||||
assert "<li>a<ul><li>nested</li></ul></li>" in markup
|
||||
parsed = parse(f"<root>{markup[markup.index('<ul'):markup.index('</body>')]}</root>")
|
||||
outer_items = parsed.find("ul").findall("li")
|
||||
assert len(outer_items) == 2
|
||||
assert outer_items[0].find("ul/li").text == "nested"
|
||||
|
||||
def test_blank_line_does_not_split_a_list(self, render_xhtmlmp):
|
||||
# A loose Markdown list (blank line between items) is still one
|
||||
# list, matching the WML renderer's own convention.
|
||||
markup = render_xhtmlmp("- a\n\n- b\n")
|
||||
assert markup.count("<ul>") == 1
|
||||
assert "<li>a</li><li>b</li>" in markup
|
||||
|
||||
def test_paragraph_between_list_items_closes_the_list(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("- a\n\nText.\n\n- b\n")
|
||||
assert markup.count("<ul>") == 2
|
||||
|
||||
def test_table(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("| A | B |\n| --- | --- |\n| 1 | 2 |\n")
|
||||
assert "<table><tr><th>A</th><th>B</th></tr><tr><td>1</td><td>2</td></tr></table>" in markup
|
||||
|
||||
def test_table_is_not_wrapped_in_a_paragraph(self, render_xhtmlmp):
|
||||
# Unlike WML, <table> is valid body-level content in XHTML-MP, so
|
||||
# no <p> wrapper is needed (or wanted).
|
||||
markup = render_xhtmlmp("| A |\n| --- |\n| 1 |\n")
|
||||
assert "<p><table" not in markup
|
||||
|
||||
|
||||
class TestInlineMapping:
|
||||
def test_bold_and_italic(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("**bold** and *italic*\n")
|
||||
assert "<strong>bold</strong>" in markup
|
||||
assert "<em>italic</em>" in markup
|
||||
|
||||
def test_code_span_becomes_code_element(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("Some `code` here.\n")
|
||||
assert "<code>code</code>" in markup
|
||||
|
||||
def test_link(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("[go](http://x.test/)\n")
|
||||
assert '<a href="http://x.test/">go</a>' in markup
|
||||
|
||||
def test_link_label_may_carry_emphasis(self, render_xhtmlmp):
|
||||
# Unlike WML, whose <a> content model is (#PCDATA | br | img)* and
|
||||
# so strips emphasis from the label, XHTML-MP's <a> allows it.
|
||||
markup = render_xhtmlmp("[**go** now](http://x.test/)\n")
|
||||
assert '<a href="http://x.test/"><strong>go</strong> now</a>' in markup
|
||||
|
||||
def test_underscore_in_url_is_not_shredded_into_emphasis(self, render_xhtmlmp):
|
||||
# Regression guard for the stash-before-emphasis ordering: without
|
||||
# it, `/a_b_c` inside the URL would be read as `_b_` -> <em>b</em>.
|
||||
markup = render_xhtmlmp("[link](/a_b_c)\n")
|
||||
assert '<a href="/a_b_c">link</a>' in markup
|
||||
|
||||
def test_asterisk_in_url_is_not_shredded_into_emphasis(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("[link](/a*b*c)\n")
|
||||
assert '<a href="/a*b*c">link</a>' in markup
|
||||
|
||||
def test_image(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp("\n")
|
||||
assert '<img src="pic.png" alt="alt text"/>' in markup
|
||||
|
||||
def test_dollar_sign_is_not_doubled(self, render_xhtmlmp):
|
||||
# XHTML-MP has none of WML's '$' -> '$$' variable-substitution
|
||||
# quirk; a literal dollar sign should pass through as one.
|
||||
markup = render_xhtmlmp("Permit fee: $5\n")
|
||||
assert "Permit fee: $5" in markup
|
||||
|
||||
def test_entities_are_escaped(self, render_xhtmlmp):
|
||||
markup = render_xhtmlmp('Fish & chips < > "\n')
|
||||
assert "Fish & chips < > "" in markup
|
||||
Loading…
Add table
Add a link
Reference in a new issue