From 3af44fb0ac88cd1c613d2fba4436a9689f62c01a Mon Sep 17 00:00:00 2001 From: randogoth Date: Sat, 26 Sep 2026 12:41:09 +0300 Subject: [PATCH] feat: add xhtmlmp renderer --- src/wapdown/cli.py | 18 ++- src/wapdown/renderers/__init__.py | 3 +- src/wapdown/renderers/xhtmlmp/__init__.py | 24 ++++ src/wapdown/renderers/xhtmlmp/inline.py | 95 +++++++++++++ src/wapdown/renderers/xhtmlmp/renderer.py | 157 +++++++++++++++++++++ tests/conftest.py | 17 ++- tests/golden/kitchen-sink-xhtmlmp.xhtml | 6 + tests/golden/trail-manual-xhtmlmp.xhtml | 6 + tests/golden/trail-rules-xhtmlmp.xhtml | 6 + tests/golden/trail-xhtmlmp.xhtml | 6 + tests/golden_cases.py | 11 ++ tests/regenerate_golden.py | 7 +- tests/test_golden.py | 16 ++- tests/test_xhtmlmp.py | 159 ++++++++++++++++++++++ 14 files changed, 524 insertions(+), 7 deletions(-) create mode 100644 src/wapdown/renderers/xhtmlmp/__init__.py create mode 100644 src/wapdown/renderers/xhtmlmp/inline.py create mode 100644 src/wapdown/renderers/xhtmlmp/renderer.py create mode 100644 tests/golden/kitchen-sink-xhtmlmp.xhtml create mode 100644 tests/golden/trail-manual-xhtmlmp.xhtml create mode 100644 tests/golden/trail-rules-xhtmlmp.xhtml create mode 100644 tests/golden/trail-xhtmlmp.xhtml create mode 100644 tests/test_xhtmlmp.py diff --git a/src/wapdown/cli.py b/src/wapdown/cli.py index eea4cfd..b3b91ce 100644 --- a/src/wapdown/cli.py +++ b/src/wapdown/cli.py @@ -19,6 +19,7 @@ from .plugins import ( register_parser, ) from .renderers import wml as _wml # noqa: F401 # registers the WML renderer plugin +from .renderers import xhtmlmp as _xhtmlmp # noqa: F401 # registers the XHTML-MP renderer plugin from .renderers.wml.renderer import DeckOutput @@ -211,8 +212,21 @@ def _cli_overrides(args: argparse.Namespace) -> Dict[str, Any]: return {dest: getattr(args, dest, None) for dest in _CONFIG_DESTS} -def write_output(output: DeckOutput, path: Optional[Path]) -> None: - """Write the deck. WML is XML served over HTTP, so lines end with LF.""" +def write_output(output: "DeckOutput | str", path: Optional[Path]) -> None: + """Write the renderer's output. WML/XHTML-MP are XML served over HTTP, + so lines end with LF. + + A plain string (as XHTML-MP produces -- it has no deck/card topology + to report) is written as-is; a DeckOutput may fan out to several files + in --deck-per-card mode. + """ + if isinstance(output, str): + if path is None: + sys.stdout.write(output) + return + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(output, encoding="utf-8") + return multi = len(output.documents) > 1 or any(doc.name for doc in output.documents) if multi: if path is None: diff --git a/src/wapdown/renderers/__init__.py b/src/wapdown/renderers/__init__.py index b7bbab1..fffa975 100644 --- a/src/wapdown/renderers/__init__.py +++ b/src/wapdown/renderers/__init__.py @@ -1,5 +1,6 @@ """Bundled renderer implementations.""" from .wml import WMLRenderer +from .xhtmlmp import XhtmlMpRenderer -__all__ = ["WMLRenderer"] +__all__ = ["WMLRenderer", "XhtmlMpRenderer"] diff --git a/src/wapdown/renderers/xhtmlmp/__init__.py b/src/wapdown/renderers/xhtmlmp/__init__.py new file mode 100644 index 0000000..26c31bb --- /dev/null +++ b/src/wapdown/renderers/xhtmlmp/__init__.py @@ -0,0 +1,24 @@ +"""XHTML-MP renderer, registered under the name `xhtmlmp`.""" + +from typing import Any + +from ...plugins import register_renderer +from .inline import escape_xhtml, process_inline +from .renderer import XHTML_PROLOG, XhtmlMpRenderer + + +def _xhtmlmp_renderer_factory(**options: Any) -> XhtmlMpRenderer: + return XhtmlMpRenderer(**options) + + +try: + register_renderer("xhtmlmp", _xhtmlmp_renderer_factory) +except ValueError: # pragma: no cover - already registered on re-import + pass + +__all__ = [ + "XHTML_PROLOG", + "XhtmlMpRenderer", + "escape_xhtml", + "process_inline", +] diff --git a/src/wapdown/renderers/xhtmlmp/inline.py b/src/wapdown/renderers/xhtmlmp/inline.py new file mode 100644 index 0000000..e473255 --- /dev/null +++ b/src/wapdown/renderers/xhtmlmp/inline.py @@ -0,0 +1,95 @@ +"""Inline Markdown -> XHTML-MP markup mapping, plus XML escaping. + +Mirrors wml/inline.py's stash-then-substitute structure and reuses its +Markdown-parsing regexes (a URL like `/a_b_c` must survive intact, which is +what the stash step protects against), but the two output grammars differ +enough that the escaping and link/code handling are not shared: + + - XHTML-MP has no '$' variable-substitution quirk. + - may contain inline markup, unlike WML's `(#PCDATA | br | img)*`, so + a link label is run through the same emphasis pass as ordinary text + instead of having its markers stripped. + - exists, so a code span becomes an element instead of plain text. +""" +from __future__ import annotations + +from typing import List + +from ..wml.inline import ( + BOLD_RE, + CODE_STASH_RE, + IMAGE_RE, + ITALIC_RE, + LINK_RE, + STRIKETHROUGH_RE, + UNDERLINE_EM_RE, + UNDERLINE_STRONG_RE, +) + + +_STASH = "\u0000{kind}{index}\u0000" +_CODE = "CODE" +_TAG = "TAG" + + +def escape_xhtml(text: str) -> str: + """XML-escape text for XHTML-MP element content or a double-quoted + attribute value. Applied once, up front, to the whole raw string before + any inline markup is turned into tags, matching wml/inline.py's + ordering, so link/image URLs captured out of the already-escaped text + are correctly escaped for attribute use too. + """ + text = text.replace("&", "&") + return text.replace("<", "<").replace(">", ">").replace('"', """) + + +def process_inline(text: str) -> str: + """Convert one run of inline Markdown into XHTML-MP markup. + + Links and images are resolved *before* the emphasis patterns run and + stashed behind sentinels, for the same reason as in wml/inline.py: a + path like `/a_b_c` or `/a*b*c` would otherwise be shredded into + tags by the emphasis regexes before LINK_RE ever saw it. + """ + escaped = escape_xhtml(text) + + code_segments: List[str] = [] + tag_segments: List[str] = [] + + def stash_code(match): + code_segments.append(match.group(0)[1:-1]) + return _STASH.format(kind=_CODE, index=len(code_segments) - 1) + + def stash(markup: str) -> str: + tag_segments.append(markup) + return _STASH.format(kind=_TAG, index=len(tag_segments) - 1) + + def stash_image(match): + alt = match.group(1).strip() + src = match.group(2).strip() + return stash(f'{alt}') + + def stash_link(match): + label = match.group(1).strip() + url = match.group(2).strip() + return stash(f'{_apply_emphasis(label)}') + + escaped = CODE_STASH_RE.sub(stash_code, escaped) + escaped = IMAGE_RE.sub(stash_image, escaped) # before LINK_RE: `![x](y)` is not a link + escaped = LINK_RE.sub(stash_link, escaped) + escaped = _apply_emphasis(escaped) + + for index, markup in enumerate(tag_segments): + escaped = escaped.replace(_STASH.format(kind=_TAG, index=index), markup) + for index, code in enumerate(code_segments): + escaped = escaped.replace(_STASH.format(kind=_CODE, index=index), f"{code}") + return escaped + + +def _apply_emphasis(text: str) -> str: + text = STRIKETHROUGH_RE.sub(lambda m: m.group(1), text) # no XHTML-MP equivalent + text = BOLD_RE.sub(lambda m: f"{m.group(1)}", text) + text = UNDERLINE_STRONG_RE.sub(lambda m: f"{m.group(1)}", text) + text = ITALIC_RE.sub(lambda m: f"{m.group(1)}", text) + text = UNDERLINE_EM_RE.sub(lambda m: f"{m.group(1)}", text) + return text diff --git a/src/wapdown/renderers/xhtmlmp/renderer.py b/src/wapdown/renderers/xhtmlmp/renderer.py new file mode 100644 index 0000000..ddaa670 --- /dev/null +++ b/src/wapdown/renderers/xhtmlmp/renderer.py @@ -0,0 +1,157 @@ +"""The XHTML-MP renderer.""" +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any, List, Optional + +from ...config import DeckConfig +from ...models import BlockEvent, BlockKind, StyleUpdateEvent +from .inline import escape_xhtml, process_inline + + +XHTML_PROLOG = ( + '\n' + '\n' +) + + +@dataclass +class _ListFrame: + """One open
    /
      , mid-render. + + `items` holds complete `
    1. ...
    2. ` strings. A deeper list is spliced + into the last item *before* its closing tag when it closes, because + XHTML nests a sub-list inside its parent
    3. , not after it -- unlike + WML, which has no list element to nest at all. + """ + + depth: int + ordered: bool + items: List[str] = field(default_factory=list) + + +class XhtmlMpRenderer: + """Render Markdown events as an XHTML-MP page for WAP 2.0 browsers and, + served as text/html, modern browsers too. + + Unlike WMLRenderer, there is no deck-size budget or card topology to + compute, so output streams incrementally per event rather than being + buffered and resolved in `finalize()`. + """ + + def __init__(self, config: Optional[DeckConfig] = None, **options: Any) -> None: + self.config = config or DeckConfig.resolve(overrides=options) + self._body: List[str] = [] + self._list_stack: List[_ListFrame] = [] + + # -- Renderer protocol --------------------------------------------------- + + def handle_event(self, event: "BlockEvent | StyleUpdateEvent") -> None: + if isinstance(event, StyleUpdateEvent): + # XHTML-MP has no per-block style stack to apply this to, by + # design; ignored like other structural mismatches. + return + handler = { + BlockKind.PARAGRAPH: self._render_paragraph, + BlockKind.HEADING: self._render_heading, + BlockKind.LIST_ITEM: self._render_list_item, + BlockKind.CODE_BLOCK: self._render_code_block, + BlockKind.BLOCKQUOTE: self._render_blockquote, + BlockKind.TABLE: self._render_table, + BlockKind.HORIZONTAL_RULE: self._render_horizontal_rule, + BlockKind.CARD_BREAK: self._close_lists, + # BLANK_LINE: a blank line between list items is loose Markdown + # formatting, not a break in the list, so it leaves any open + # list open -- matching the WML renderer's own convention of + # not resetting list state on a blank line. + }.get(event.kind) + if handler is not None: + handler(event.payload) + + def finalize(self) -> str: + self._close_lists() + title = escape_xhtml(self.config.title) if self.config.title else "" + head = f"{title}" + body = "".join(self._body) + return ( + f"{XHTML_PROLOG}" + '\n' + f"{head}\n" + f"{body}\n" + "\n" + ) + + # -- List nesting --------------------------------------------------------- + + def _render_list_item(self, payload) -> None: + indent = payload.indent.replace("\t", " ") + depth = max(0, len(indent) // 2) + text = self._inline(payload.text) + self._open_list(depth, payload.ordered) + self._list_stack[-1].items.append(f"
    4. {text}
    5. ") + + def _open_list(self, depth: int, ordered: bool) -> None: + while self._list_stack and ( + self._list_stack[-1].depth > depth + or (self._list_stack[-1].depth == depth and self._list_stack[-1].ordered != ordered) + ): + self._close_one_list() + if not self._list_stack or self._list_stack[-1].depth < depth: + self._list_stack.append(_ListFrame(depth, ordered)) + + def _close_one_list(self) -> None: + frame = self._list_stack.pop() + tag = "ol" if frame.ordered else "ul" + wrapped = f"<{tag}>{''.join(frame.items)}" + if self._list_stack: + parent_items = self._list_stack[-1].items + parent_items[-1] = parent_items[-1][: -len("")] + wrapped + "" + else: + self._body.append(wrapped) + + def _close_lists(self, *_: object) -> None: + while self._list_stack: + self._close_one_list() + + # -- Block markup mapping ------------------------------------------------- + + def _render_paragraph(self, payload) -> None: + self._close_lists() + self._body.append(f"

      {self._inline(payload.text)}

      ") + + def _render_heading(self, payload) -> None: + self._close_lists() + level = max(1, min(6, payload.level)) + text = self._inline(payload.text) + self._body.append(f"{text}") + + def _render_code_block(self, payload) -> None: + self._close_lists() + lines = "\n".join(escape_xhtml(line.rstrip("\n")) for line in payload.lines) + self._body.append(f"
      {lines}
      ") + + def _render_blockquote(self, payload) -> None: + self._close_lists() + self._body.append(f"

      {self._inline(payload.text)}

      ") + + def _render_table(self, payload) -> None: + self._close_lists() + rows = payload.rows + if not rows: + return + columns = max(len(row) for row in rows) + tr_parts: List[str] = [] + for row_index, row in enumerate(rows): + cells = list(row) + [""] * (columns - len(row)) + cell_tag = "th" if row_index == 0 else "td" + tds = "".join(f"<{cell_tag}>{self._inline(cell)}" for cell in cells) + tr_parts.append(f"{tds}") + self._body.append(f"{''.join(tr_parts)}
      ") + + def _render_horizontal_rule(self, _payload) -> None: + self._close_lists() + self._body.append("
      ") + + def _inline(self, text: str) -> str: + return process_inline(text) diff --git a/tests/conftest.py b/tests/conftest.py index a8cbcad..cc0c44a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -25,11 +25,11 @@ def render_source(text: str, **options: Any) -> str: return convert(body, config=config, base_path=EXAMPLES).text -def render_file(path: Path, **options: Any) -> DeckOutput: +def render_file(path: Path, *, renderer_name: str = "wml", **options: Any): lines = read_lines(path) frontmatter, body = strip_frontmatter(lines) config = DeckConfig.resolve(frontmatter, options) - return convert(body, config=config, base_path=path.parent) + return convert(body, config=config, base_path=path.parent, renderer_name=renderer_name) @pytest.fixture @@ -37,6 +37,19 @@ def render(): return render_source +def render_xhtmlmp_source(text: str, **options: Any) -> str: + """Like `render_source`, but through the XHTML-MP renderer.""" + lines = text.splitlines(keepends=True) + frontmatter, body = strip_frontmatter(lines) + config = DeckConfig.resolve(frontmatter, options) + return convert(body, config=config, base_path=EXAMPLES, renderer_name="xhtmlmp") + + +@pytest.fixture +def render_xhtmlmp(): + return render_xhtmlmp_source + + @pytest.fixture def render_output(): """Like `render`, but exposes every output file (deck-per-card mode).""" diff --git a/tests/golden/kitchen-sink-xhtmlmp.xhtml b/tests/golden/kitchen-sink-xhtmlmp.xhtml new file mode 100644 index 0000000..b33ef32 --- /dev/null +++ b/tests/golden/kitchen-sink-xhtmlmp.xhtml @@ -0,0 +1,6 @@ + + + +Kitchen Sink +

      Kitchen Sink

      A deck exercising every block and inline construct wapdown understands.

      Links and emphasis

      Tricky URL: release notes and a starred path build log.

      Bold, italic, inline code, struck, and a literal $5 fee.

      A table

      TrailStatusFee
      North Loopopen$2
      Summit Spuricy$5

      Other blocks

      A quoted warning.

      1. First
      2. Second

      indented code $HOME
      + diff --git a/tests/golden/trail-manual-xhtmlmp.xhtml b/tests/golden/trail-manual-xhtmlmp.xhtml new file mode 100644 index 0000000..ce6bef4 --- /dev/null +++ b/tests/golden/trail-manual-xhtmlmp.xhtml @@ -0,0 +1,6 @@ + + + + +

      Live updates for the ridge trail network. Reception is spotty past the tree line, check before you go.

      Cold and clear. Wind: 15 mph gusting from the northwest. Permit fee: $5.

      • North Loop: open
      • South Loop: closed
      • Summit Spur: open, ice above 2000m

      Ranger station: call dispatch

      Trail map

      + diff --git a/tests/golden/trail-rules-xhtmlmp.xhtml b/tests/golden/trail-rules-xhtmlmp.xhtml new file mode 100644 index 0000000..800b5f1 --- /dev/null +++ b/tests/golden/trail-rules-xhtmlmp.xhtml @@ -0,0 +1,6 @@ + + + + +

      Live updates for the ridge trail network. Reception is spotty past the tree line, check before you go.


      Weather

      Cold and clear. Wind: 15 mph gusting from the northwest. Permit fee: $5.


      Trail Status

      • North Loop: open
      • South Loop: closed
      • Summit Spur: open, ice above 2000m

      Contact

      Ranger station: call dispatch


      + diff --git a/tests/golden/trail-xhtmlmp.xhtml b/tests/golden/trail-xhtmlmp.xhtml new file mode 100644 index 0000000..85fcf78 --- /dev/null +++ b/tests/golden/trail-xhtmlmp.xhtml @@ -0,0 +1,6 @@ + + + + +

      Trail Conditions

      Live updates for the ridge trail network. Reception is spotty past the tree line, check before you go.

      Weather

      Cold and clear. Wind: 15 mph gusting from the northwest. Permit fee: $5.

      Trail Status

      • North Loop: open
      • South Loop: closed
      • Summit Spur: open, ice above 2000m

      Contact

      Ranger station: call dispatch

      Trail map

      + diff --git a/tests/golden_cases.py b/tests/golden_cases.py index 74f849d..1256853 100644 --- a/tests/golden_cases.py +++ b/tests/golden_cases.py @@ -30,3 +30,14 @@ CASES: Dict[str, Tuple[str, Dict[str, Any]]] = { MULTI_CASES: Dict[str, Tuple[str, Dict[str, Any]]] = { "trail-decks": ("trail-manual.md", {"deck_per_card": True, "home_label": "Home"}), } + +# name -> (example filename, renderer options), rendered through the +# XHTML-MP renderer. A separate table from CASES because XHTML-MP output is +# a plain string, not a DeckOutput -- there is no deck/card topology to +# vary, so only options that affect ordinary block/inline rendering apply. +XHTMLMP_CASES: Dict[str, Tuple[str, Dict[str, Any]]] = { + "trail-xhtmlmp": ("trail.md", {}), + "trail-manual-xhtmlmp": ("trail-manual.md", {}), + "trail-rules-xhtmlmp": ("trail-rules.md", {}), + "kitchen-sink-xhtmlmp": ("kitchen-sink.md", {}), +} diff --git a/tests/regenerate_golden.py b/tests/regenerate_golden.py index df5c5c2..3755d55 100644 --- a/tests/regenerate_golden.py +++ b/tests/regenerate_golden.py @@ -16,7 +16,7 @@ from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from conftest import EXAMPLES, GOLDEN, render_file # noqa: E402 -from golden_cases import CASES, MULTI_CASES # noqa: E402 +from golden_cases import CASES, MULTI_CASES, XHTMLMP_CASES # noqa: E402 def main() -> int: @@ -37,6 +37,11 @@ def main() -> int: document.text, encoding="utf-8" ) print(f"wrote {name}/ ({len(list(directory.iterdir()))} files)") + + for name, (source, options) in XHTMLMP_CASES.items(): + output = render_file(EXAMPLES / source, renderer_name="xhtmlmp", **options) + (GOLDEN / f"{name}.xhtml").write_text(output, encoding="utf-8") + print(f"wrote {name}.xhtml") return 0 diff --git a/tests/test_golden.py b/tests/test_golden.py index 02e2fb4..b06b3e3 100644 --- a/tests/test_golden.py +++ b/tests/test_golden.py @@ -11,7 +11,7 @@ from xml.etree import ElementTree import pytest from conftest import EXAMPLES, GOLDEN, render_file -from golden_cases import CASES, MULTI_CASES +from golden_cases import CASES, MULTI_CASES, XHTMLMP_CASES REGENERATE_HINT = ( @@ -49,6 +49,20 @@ def test_golden_files_are_well_formed(path): ElementTree.fromstring(path.read_text(encoding="utf-8")) +@pytest.mark.parametrize("name", sorted(XHTMLMP_CASES)) +def test_matches_golden_xhtmlmp(name): + source, options = XHTMLMP_CASES[name] + golden = GOLDEN / f"{name}.xhtml" + assert golden.exists(), f"missing golden for '{name}'; {REGENERATE_HINT}" + actual = render_file(EXAMPLES / source, renderer_name="xhtmlmp", **options) + assert actual == golden.read_text(encoding="utf-8"), REGENERATE_HINT + + +@pytest.mark.parametrize("path", sorted(GOLDEN.glob("**/*.xhtml")), ids=lambda p: p.name) +def test_xhtmlmp_golden_files_are_well_formed(path): + ElementTree.fromstring(path.read_text(encoding="utf-8")) + + def test_every_example_has_a_case(): """A new examples/*.md without a golden case would go unverified.""" covered = {source for source, _ in list(CASES.values()) + list(MULTI_CASES.values())} diff --git a/tests/test_xhtmlmp.py b/tests/test_xhtmlmp.py new file mode 100644 index 0000000..6905a3a --- /dev/null +++ b/tests/test_xhtmlmp.py @@ -0,0 +1,159 @@ +"""XHTML-MP emission: prolog, head, block mapping, and well-formedness.""" +from __future__ import annotations + +from xml.etree import ElementTree + +import pytest + +from wapdown.renderers.xhtmlmp.renderer import XHTML_PROLOG + + +TWO_SECTIONS = "Intro.\n\n{.card One}\nFirst.\n\n{.card Two}\nSecond.\n" + + +def parse(markup: str) -> ElementTree.Element: + """Parse the page, proving it is well-formed XML. + + ElementTree honours the DOCTYPE declaration without trying to fetch the + external DTD, so this never depends on a network round trip to + wapforum.org. + """ + return ElementTree.fromstring(markup) + + +class TestDocumentShape: + def test_prolog_and_doctype(self, render_xhtmlmp): + markup = render_xhtmlmp("Body.\n") + assert markup.startswith(XHTML_PROLOG) + assert 'PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.0//EN"' in markup + + def test_root_element_is_html(self, render_xhtmlmp): + # ElementTree namespace-qualifies the tag because of the xhtml xmlns. + assert parse(render_xhtmlmp("Body.\n")).tag == "{http://www.w3.org/1999/xhtml}html" + + def test_output_is_well_formed(self, render_xhtmlmp): + parse(render_xhtmlmp(TWO_SECTIONS)) + + def test_no_carriage_returns(self, render_xhtmlmp): + assert "\r" not in render_xhtmlmp(TWO_SECTIONS) + + def test_card_breaks_do_not_split_the_page(self, render_xhtmlmp): + # Cards are a WAP 1.x screen-budget concern; XHTML-MP renders one + # continuous page and a {.card} marker draws nothing of its own. + markup = render_xhtmlmp(TWO_SECTIONS) + assert markup.count("Fish & Chips" in markup + + +class TestBlockMapping: + @pytest.mark.parametrize( + ("source", "expected"), + [ + ("Just text.\n", "

      Just text.

      "), + ("> quoted\n", "

      quoted

      "), + ("```\ncode\n```\n", "
      code
      "), + ("---\n", "
      "), + ], + ) + def test_blocks(self, render_xhtmlmp, source, expected): + assert expected in render_xhtmlmp(source) + + @pytest.mark.parametrize("level", [1, 2, 3, 4, 5, 6]) + def test_headings_use_real_heading_elements(self, render_xhtmlmp, level): + markup = render_xhtmlmp(f"{'#' * level} Title\n") + assert f"Title" in markup + assert "" not in markup # no FIGlet/WML-style banner substitute + + def test_heading_level_beyond_six_clamps(self, render_xhtmlmp): + # Markdown itself caps at h6, but a manually-crafted payload + # shouldn't be able to emit an invalid element name. + markup = render_xhtmlmp("###### Deep\n") + assert "
      Deep
      " in markup + + def test_unordered_list(self, render_xhtmlmp): + markup = render_xhtmlmp("- a\n- b\n") + assert "
      • a
      • b
      " in markup + + def test_ordered_list(self, render_xhtmlmp): + markup = render_xhtmlmp("1. a\n2. b\n") + assert "
      1. a
      2. b
      " in markup + + def test_nested_list_is_spliced_into_parent_item(self, render_xhtmlmp): + markup = render_xhtmlmp("- a\n - nested\n- b\n") + # The sub-list must sit *inside* its parent
    6. , not after it, or + # the result is not a valid nested list. + assert "
    7. a
      • nested
    8. " in markup + parsed = parse(f"{markup[markup.index('')]}") + outer_items = parsed.find("ul").findall("li") + assert len(outer_items) == 2 + assert outer_items[0].find("ul/li").text == "nested" + + def test_blank_line_does_not_split_a_list(self, render_xhtmlmp): + # A loose Markdown list (blank line between items) is still one + # list, matching the WML renderer's own convention. + markup = render_xhtmlmp("- a\n\n- b\n") + assert markup.count("
        ") == 1 + assert "
      • a
      • b
      • " in markup + + def test_paragraph_between_list_items_closes_the_list(self, render_xhtmlmp): + markup = render_xhtmlmp("- a\n\nText.\n\n- b\n") + assert markup.count("
          ") == 2 + + def test_table(self, render_xhtmlmp): + markup = render_xhtmlmp("| A | B |\n| --- | --- |\n| 1 | 2 |\n") + assert "
          AB
          12
          " in markup + + def test_table_is_not_wrapped_in_a_paragraph(self, render_xhtmlmp): + # Unlike WML, is valid body-level content in XHTML-MP, so + # no

          wrapper is needed (or wanted). + markup = render_xhtmlmp("| A |\n| --- |\n| 1 |\n") + assert "

          bold" in markup + assert "italic" in markup + + def test_code_span_becomes_code_element(self, render_xhtmlmp): + markup = render_xhtmlmp("Some `code` here.\n") + assert "code" in markup + + def test_link(self, render_xhtmlmp): + markup = render_xhtmlmp("[go](http://x.test/)\n") + assert 'go' in markup + + def test_link_label_may_carry_emphasis(self, render_xhtmlmp): + # Unlike WML, whose content model is (#PCDATA | br | img)* and + # so strips emphasis from the label, XHTML-MP's allows it. + markup = render_xhtmlmp("[**go** now](http://x.test/)\n") + assert 'go now' in markup + + def test_underscore_in_url_is_not_shredded_into_emphasis(self, render_xhtmlmp): + # Regression guard for the stash-before-emphasis ordering: without + # it, `/a_b_c` inside the URL would be read as `_b_` -> b. + markup = render_xhtmlmp("[link](/a_b_c)\n") + assert 'link' in markup + + def test_asterisk_in_url_is_not_shredded_into_emphasis(self, render_xhtmlmp): + markup = render_xhtmlmp("[link](/a*b*c)\n") + assert 'link' in markup + + def test_image(self, render_xhtmlmp): + markup = render_xhtmlmp("![alt text](pic.png)\n") + assert 'alt text' in markup + + def test_dollar_sign_is_not_doubled(self, render_xhtmlmp): + # XHTML-MP has none of WML's '$' -> '$$' variable-substitution + # quirk; a literal dollar sign should pass through as one. + markup = render_xhtmlmp("Permit fee: $5\n") + assert "Permit fee: $5" in markup + + def test_entities_are_escaped(self, render_xhtmlmp): + markup = render_xhtmlmp('Fish & chips < > "\n') + assert "Fish & chips < > "" in markup