Extract the WML exporter from md2txt into wapdown

md2txt is a suite of minimal text-based markup exporters, and every other
renderer in it (text, micron, ama, gemini, nex) is a straight transform of
a document into a flat text format. The WML renderer is neither: it emits
an XML tree, and it restructures the document rather than transforming it
-- splitting one file into many <card>s, inventing navigation between
them, building a menu card that exists in no source document, and
paginating on a byte budget. That is why it needed levers no sibling
renderer needed, and why it alone forced a CARD_BREAK block kind and a
{.card} directive into the shared parser that every other renderer
ignores.

wapdown is that renderer given a repository of its own, where
restructuring a document into a navigable deck is the point.

Ported: the Markdown parser, block event models, and pipeline core, each
trimmed to what a deck compiler actually uses (no FIGlet, no hyphenation,
no ASCII-art includes, no text-layout frontmatter), plus the eight inline
regexes the renderer had been borrowing from a 1039-line text renderer.
The plugin registry comes along so other WAP-era targets can register
alongside wml later. Zero runtime dependencies.

Config is now first-class: every lever is an argparse flag with
validation and a frontmatter equivalent, layered CLI > frontmatter >
default, replacing the generic --renderer-option KEY=VALUE passthrough.

Five defects found while reading the code, each with a regression test:

- Link and image URLs were shredded by the emphasis patterns, which ran
  before LINK_RE: a path like /v1_2_3/ became /v1<i>2</i>3/ inside the
  href. Links and images are now resolved and stashed first.
- <table> was emitted as a direct child of <card>. The WML 1.3 content
  model is (onevent*, timer?, (do | p | pre)*), so it must sit in a <p>.
- Link labels carried emphasis, but <a> is declared (#PCDATA | br | img)*
  -- there is nowhere valid for a <b> inside one. Markers are now stripped.
- The menu card was built directly and never byte-packed, so the one card
  most likely to be large was the one card exempt from the budget.
- Output was written with CRLF line endings. WML is XML served over HTTP.

New, WAP-idiomatic rather than Markdown-idiomatic:

- deck-level <template> hoisting nav that would otherwise repeat per card
- <select> menus, picked with a keypad digit instead of scrolled to
- --deck-per-card, one file per section cross-linked by filename, which
  is the real answer to per-deck cache limits
- <head> metadata: Cache-Control max-age and <access>
- a Home options softkey, which <prev/> alone cannot provide
- an image policy (keep/alt/drop) for the WBMP-only reality of WAP 1.x
  browsers -- no conversion is performed

186 tests, where neither repo had any. Golden files pin whole-document
output; every deck is validated against the real WML 1.3 DTD, vendored at
tests/dtd/wml13.dtd. Rendering examples/trail.md still matches md2txt's
committed output byte for byte, modulo the CRLF fix.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
randogoth 2026-09-23 09:27:49 +03:00
commit b83e24e2f7
51 changed files with 4016 additions and 0 deletions

6
src/wapdown/__init__.py Normal file
View file

@ -0,0 +1,6 @@
"""Public package interface for wapdown."""
from .cli import convert, main
from .config import DeckConfig
__all__ = ["DeckConfig", "convert", "main"]

7
src/wapdown/__main__.py Normal file
View file

@ -0,0 +1,7 @@
"""Entry point for `python -m wapdown`."""
from .cli import main
if __name__ == "__main__":
raise SystemExit(main())

250
src/wapdown/cli.py Normal file
View file

@ -0,0 +1,250 @@
#!/usr/bin/env python3
"""Compile a Markdown document into a WML deck for WAP browsers."""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from typing import Any, Dict, Iterable, List, Optional
from .config import IMAGE_POLICIES, MENU_STYLES, ConfigError, DeckConfig
from .conversion.core import read_lines, run_conversion, strip_frontmatter
from .models import BlockStyle
from .parsers.markdown import MarkdownParser
from .plugins import (
available_parsers,
available_renderers,
get_parser_factory,
get_renderer_factory,
register_parser,
)
from .renderers import wml as _wml # noqa: F401 # registers the WML renderer plugin
from .renderers.wml.renderer import DeckOutput
def _markdown_parser_factory(*, base_style: BlockStyle, **_: Any) -> MarkdownParser:
return MarkdownParser(base_style)
try:
register_parser("markdown", _markdown_parser_factory)
except ValueError: # pragma: no cover - already registered on re-import
pass
def convert(
lines: Iterable[str],
*,
config: DeckConfig,
base_path: Optional[Path] = None,
parser_name: str = "markdown",
renderer_name: str = "wml",
) -> DeckOutput:
return run_conversion(
lines,
parser_factory=get_parser_factory(parser_name),
renderer_factory=get_renderer_factory(renderer_name),
renderer_options={"config": config},
base_path=base_path,
)
def build_parser() -> argparse.ArgumentParser:
# Every option defaults to None so that "absent" stays distinguishable
# from "explicitly set to the default"; DeckConfig.resolve then layers
# CLI over frontmatter over defaults.
parser = argparse.ArgumentParser(
prog="wapdown",
description="Compile Markdown into a WML deck for WAP browsers.",
epilog=(
"Every option can also be set in the document's YAML-ish frontmatter "
"using the long flag name with underscores (--split-level -> split_level). "
"Command-line flags win over frontmatter."
),
)
parser.add_argument("input_path", type=Path, help="Path to the Markdown input file.")
parser.add_argument(
"-o",
"--output",
type=Path,
help="File to write the deck to (a directory with --deck-per-card). Defaults to stdout.",
)
structure = parser.add_argument_group("deck structure")
structure.add_argument("--title", help="Overall deck title, used on the menu card.")
structure.add_argument(
"--split-level",
type=int,
metavar="N",
help="Start a new card at every level-N heading (1-6; 0 disables). "
"Ignored when the document contains {.card} markers.",
)
structure.add_argument(
"--max-card-bytes",
type=int,
metavar="N",
help="Byte safety net per card; 0 disables (default: 1400).",
)
menu_group = structure.add_mutually_exclusive_group()
menu_group.add_argument(
"--menu",
dest="menu",
action="store_true",
default=None,
help="Build a hub-and-spoke menu card (the default for multi-card decks).",
)
menu_group.add_argument(
"--no-menu",
dest="menu",
action="store_false",
default=None,
help="Chain cards linearly with Next/Back instead of building a menu.",
)
structure.add_argument(
"--menu-style",
choices=MENU_STYLES,
help="Render menu choices as anchors or as a keypad-pickable <select> (default: links).",
)
structure.add_argument(
"--deck-per-card",
dest="deck_per_card",
action="store_true",
default=None,
help="Write one .wml file per section into the output directory, cross-linked by filename.",
)
nav = parser.add_argument_group("navigation")
nav.add_argument("--nav-next-label", metavar="TEXT", help="Forward label (default: More).")
nav.add_argument("--nav-prev-label", metavar="TEXT", help="Label between overflow parts (default: Prev).")
nav.add_argument("--nav-back-label", metavar="TEXT", help="Label back to the menu (default: Back).")
nav.add_argument(
"--home-label",
metavar="TEXT",
help="Add an options softkey jumping straight to the menu card.",
)
template_group = nav.add_mutually_exclusive_group()
template_group.add_argument(
"--template-nav",
dest="template_nav",
action="store_true",
default=None,
help="Hoist shared nav into a deck-level <template> (the default).",
)
template_group.add_argument(
"--no-template-nav",
dest="template_nav",
action="store_false",
default=None,
help="Repeat nav markup on every card instead.",
)
content = parser.add_argument_group("content")
content.add_argument(
"--images",
choices=IMAGE_POLICIES,
help="What to do with non-WBMP images: keep them, replace with alt text, or drop (default: keep). "
"No format conversion is ever performed.",
)
head = parser.add_argument_group("deck head")
head.add_argument(
"--cache-control",
type=int,
metavar="SECONDS",
help="Emit a Cache-Control max-age meta element.",
)
head.add_argument("--access-domain", metavar="DOMAIN", help="Emit <access domain=...>.")
head.add_argument("--access-path", metavar="PATH", help="Emit <access path=...>.")
plugins = parser.add_argument_group("plugins")
plugins.add_argument(
"--parser",
default="markdown",
choices=available_parsers() or ["markdown"],
help="Name of the parser plugin to use.",
)
plugins.add_argument(
"--renderer",
default="wml",
choices=available_renderers() or ["wml"],
help="Name of the renderer plugin to use.",
)
return parser
_CONFIG_DESTS = (
"title",
"split_level",
"max_card_bytes",
"menu",
"menu_style",
"deck_per_card",
"nav_next_label",
"nav_prev_label",
"nav_back_label",
"home_label",
"template_nav",
"images",
"cache_control",
"access_domain",
"access_path",
)
def _cli_overrides(args: argparse.Namespace) -> Dict[str, Any]:
return {dest: getattr(args, dest, None) for dest in _CONFIG_DESTS}
def write_output(output: DeckOutput, path: Optional[Path]) -> None:
"""Write the deck. WML is XML served over HTTP, so lines end with LF."""
multi = len(output.documents) > 1 or any(doc.name for doc in output.documents)
if multi:
if path is None:
raise SystemExit(
"wapdown: --deck-per-card writes several files; pass -o DIRECTORY."
)
path.mkdir(parents=True, exist_ok=True)
for doc in output.documents:
(path / (doc.name or "index.wml")).write_text(doc.text, encoding="utf-8")
return
text = output.text
if path is None:
sys.stdout.write(text)
return
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text, encoding="utf-8")
def main(argv: Optional[List[str]] = None) -> int:
args = build_parser().parse_args(argv)
try:
lines = read_lines(args.input_path)
except OSError as exc:
sys.stderr.write(f"wapdown: {exc}\n")
return 2
frontmatter, content = strip_frontmatter(lines)
try:
config = DeckConfig.resolve(frontmatter, _cli_overrides(args))
except ConfigError as exc:
sys.stderr.write(f"wapdown: {exc}\n")
return 2
try:
output = convert(
content,
config=config,
base_path=args.input_path.parent,
parser_name=args.parser,
renderer_name=args.renderer,
)
except (KeyError, TypeError, FileNotFoundError, RuntimeError) as exc:
sys.stderr.write(f"wapdown: {exc}\n")
return 2
write_output(output, args.output)
return 0
if __name__ == "__main__":
raise SystemExit(main())

130
src/wapdown/config.py Normal file
View file

@ -0,0 +1,130 @@
"""Deck configuration: defaults, document frontmatter, and CLI overrides.
Precedence is CLI flag > frontmatter > default. The CLI achieves that by
giving every option `default=None`, so "not given on the command line"
stays distinguishable from "given a value that happens to equal the
default"; `DeckConfig.resolve` then layers the three sources.
Frontmatter keys are the long flag names with underscores
(`--split-level` <-> `split_level`), so one table documents both.
"""
from __future__ import annotations
import sys
from dataclasses import dataclass, fields
from typing import Any, Dict, Mapping, Optional
MENU_STYLES = ("links", "select")
IMAGE_POLICIES = ("keep", "alt", "drop")
# A conservative nod to classic WAP 1.x per-deck cache limits; purely a
# safety net for a section that's still too big after section-based
# splitting, not the primary splitting mechanism.
DEFAULT_MAX_CARD_BYTES = 1400
class ConfigError(ValueError):
"""Raised for an out-of-range or unparseable configuration value."""
@dataclass
class DeckConfig:
# Structure
title: Optional[str] = None
split_level: int = 0 # 0 = heading splitting disabled
max_card_bytes: int = DEFAULT_MAX_CARD_BYTES # 0 = disabled
menu: bool = True
menu_style: str = "links"
deck_per_card: bool = False
# Navigation
nav_next_label: str = "More"
nav_prev_label: str = "Prev"
nav_back_label: str = "Back"
home_label: Optional[str] = None
template_nav: bool = True
# Content
images: str = "keep"
# Deck <head>
cache_control: Optional[int] = None
access_domain: Optional[str] = None
access_path: Optional[str] = None
@classmethod
def resolve(
cls,
frontmatter: Optional[Mapping[str, str]] = None,
overrides: Optional[Mapping[str, Any]] = None,
) -> "DeckConfig":
config = cls()
known = {f.name for f in fields(cls)}
for key, raw in (frontmatter or {}).items():
name = key.strip().replace("-", "_")
if name not in known:
print(
f"wapdown: warning: unknown frontmatter key '{key}' ignored",
file=sys.stderr,
)
continue
setattr(config, name, _coerce(name, raw))
for name, value in (overrides or {}).items():
if value is None or name not in known:
continue
setattr(config, name, _coerce(name, value))
config.validate()
return config
def validate(self) -> None:
if not 0 <= self.split_level <= 6:
raise ConfigError(f"split_level must be between 0 and 6, got {self.split_level}")
if self.max_card_bytes < 0:
raise ConfigError(f"max_card_bytes must be 0 or greater, got {self.max_card_bytes}")
if self.menu_style not in MENU_STYLES:
raise ConfigError(
f"menu_style must be one of {', '.join(MENU_STYLES)}, got '{self.menu_style}'"
)
if self.images not in IMAGE_POLICIES:
raise ConfigError(
f"images must be one of {', '.join(IMAGE_POLICIES)}, got '{self.images}'"
)
if self.cache_control is not None and self.cache_control < 0:
raise ConfigError(f"cache_control must be 0 or greater, got {self.cache_control}")
def as_renderer_options(self) -> Dict[str, Any]:
return {f.name: getattr(self, f.name) for f in fields(self)}
_BOOL_FIELDS = {"menu", "template_nav", "deck_per_card"}
_INT_FIELDS = {"split_level", "max_card_bytes", "cache_control"}
_TRUE = {"true", "yes", "1", "on"}
_FALSE = {"false", "no", "0", "off"}
def _coerce(name: str, value: Any) -> Any:
"""Normalize a frontmatter string (or an already-typed CLI value)."""
if name in _BOOL_FIELDS:
if isinstance(value, bool):
return value
lowered = str(value).strip().lower()
if lowered in _TRUE:
return True
if lowered in _FALSE:
return False
raise ConfigError(f"{name} must be a boolean, got '{value}'")
if name in _INT_FIELDS:
if isinstance(value, bool):
raise ConfigError(f"{name} must be a number, got '{value}'")
if isinstance(value, int):
return value
text = str(value).strip()
try:
return int(text)
except ValueError as exc:
raise ConfigError(f"{name} must be a number, got '{value}'") from exc
text = str(value).strip()
if name in {"menu_style", "images"}:
return text.lower()
return text or None if name in {"title", "home_label", "access_domain", "access_path"} else text

View file

@ -0,0 +1,19 @@
"""Conversion pipeline helpers."""
from .core import (
ParserFactory,
RendererFactory,
read_lines,
run_conversion,
run_pipeline,
strip_frontmatter,
)
__all__ = [
"ParserFactory",
"RendererFactory",
"read_lines",
"run_conversion",
"run_pipeline",
"strip_frontmatter",
]

View file

@ -0,0 +1,200 @@
from __future__ import annotations
import json
import re
from pathlib import Path
from typing import Any, Dict, Iterable, Iterator, List, Optional, Protocol, Set, Tuple, TypeVar
from ..models import BlockEvent, BlockStyle, StyleUpdateEvent
INCLUDE_WIKILINK_PATTERN = re.compile(r"^\s*!\[\[(.+?)\]\]\s*$")
INCLUDE_DIRECTIVE_PATTERN = re.compile(r"^\s*\{\s*\.include\s+(.+?)\s*\}\s*$")
TABLE_SENTINEL_PREFIX = "\u0000TABLE:"
SEPARATOR_CELL_PATTERN = re.compile(r"^:?-{1,}:?$")
Event = BlockEvent | StyleUpdateEvent
RendererOutput = TypeVar("RendererOutput")
class Parser(Protocol):
def parse(self, lines: Iterable[str]) -> Iterator[Event]:
...
class Renderer(Protocol[RendererOutput]):
def handle_event(self, event: Event) -> None:
...
def finalize(self) -> RendererOutput:
...
class ParserFactory(Protocol):
def __call__(self, *, base_style: BlockStyle, **kwargs: Any) -> Parser:
...
class RendererFactory(Protocol[RendererOutput]):
def __call__(self, **kwargs: Any) -> Renderer[RendererOutput]:
...
def read_lines(path: Path) -> List[str]:
with path.open("r", encoding="utf-8") as handle:
return handle.readlines()
def expand_includes(
lines: List[str],
base_dir: Path,
include_stack: Set[Path],
) -> List[str]:
expanded: List[str] = []
for line in lines:
target = _extract_include_target(line)
if target is None:
expanded.append(line)
continue
target_path = (base_dir / target).resolve()
if target_path in include_stack:
raise RuntimeError(f"Circular include detected for '{target_path}'.")
if not target_path.exists():
raise FileNotFoundError(f"Included file '{target_path}' was not found.")
include_stack.add(target_path)
included_lines = read_lines(target_path)
_, include_body = strip_frontmatter(included_lines)
included_content = expand_includes(include_body, target_path.parent, include_stack)
expanded.extend(included_content)
include_stack.remove(target_path)
return expanded
def strip_frontmatter(lines: List[str]) -> Tuple[Dict[str, str], List[str]]:
"""Split a leading `---` fenced block off the document.
Returns the raw key/value pairs and the remaining body. Interpreting
those pairs is `wapdown.config`'s job, not the pipeline's; an included
file's frontmatter is parsed only so it can be discarded.
"""
if not lines or lines[0].strip() != "---":
return {}, lines
fields: Dict[str, str] = {}
idx = 1
while idx < len(lines):
if lines[idx].strip() == "---":
break
if ":" in lines[idx]:
key, value = lines[idx].split(":", 1)
fields[key.strip()] = value.strip()
idx += 1
if idx >= len(lines):
# Unterminated fence: treat the whole thing as body.
return {}, lines
remaining = lines[idx + 1 :] if idx + 1 < len(lines) else []
return fields, remaining
def _extract_include_target(line: str) -> Optional[str]:
stripped = line.rstrip("\n")
match = INCLUDE_WIKILINK_PATTERN.match(stripped)
if match:
return _normalize_include_target(match.group(1))
match = INCLUDE_DIRECTIVE_PATTERN.match(stripped)
if match:
return _normalize_include_target(match.group(1))
return None
def _normalize_include_target(value: str) -> str:
trimmed = value.strip()
if len(trimmed) >= 2 and ((trimmed[0] == trimmed[-1]) and trimmed[0] in {"'", '"'}):
trimmed = trimmed[1:-1].strip()
return trimmed
def _split_pipe_row(line: str) -> List[str]:
trimmed = line.strip()
if trimmed.startswith("|"):
trimmed = trimmed[1:]
if trimmed.endswith("|"):
trimmed = trimmed[:-1]
return [cell.strip() for cell in trimmed.split("|")]
def _is_table_separator_row(line: str, expected_cells: int) -> bool:
stripped = line.strip()
if not stripped:
return False
cells = _split_pipe_row(stripped)
if len(cells) != expected_cells:
return False
return all(SEPARATOR_CELL_PATTERN.match(cell) for cell in cells)
def expand_tables(lines: List[str]) -> List[str]:
"""Collapse GFM-style pipe tables into single JSON sentinel lines.
The parser only ever sees one physical line per table, so no
lookahead/pushback is needed in its line-by-line loop. Runs after
`expand_includes` so tables inside included files are also detected.
"""
expanded: List[str] = []
index = 0
total = len(lines)
while index < total:
line = lines[index]
stripped = line.rstrip("\n")
if stripped.startswith(TABLE_SENTINEL_PREFIX):
expanded.append(line)
index += 1
continue
if "|" in stripped and stripped.strip() and index + 1 < total:
header_cells = _split_pipe_row(stripped)
if header_cells and _is_table_separator_row(lines[index + 1].rstrip("\n"), len(header_cells)):
rows = [header_cells]
index += 2 # header + separator row
while index < total:
row_line = lines[index].rstrip("\n")
if not row_line.strip() or "|" not in row_line:
break
rows.append(_split_pipe_row(row_line))
index += 1
sentinel = f"{TABLE_SENTINEL_PREFIX}{json.dumps({'rows': rows})}\n"
expanded.append(sentinel)
continue
expanded.append(line)
index += 1
return expanded
def run_pipeline(
lines: Iterable[str],
*,
parser: Parser,
renderer: Renderer[RendererOutput],
base_path: Optional[Path] = None,
) -> RendererOutput:
base_dir = (base_path or Path.cwd()).resolve()
expanded_lines = expand_includes(list(lines), base_dir, set())
expanded_lines = expand_tables(expanded_lines)
for event in parser.parse(expanded_lines):
renderer.handle_event(event)
return renderer.finalize()
def run_conversion(
lines: Iterable[str],
*,
parser_factory: ParserFactory,
renderer_factory: RendererFactory[RendererOutput],
parser_options: Optional[Dict[str, Any]] = None,
renderer_options: Optional[Dict[str, Any]] = None,
base_path: Optional[Path] = None,
) -> RendererOutput:
parser = parser_factory(
base_style=BlockStyle(align="left", margin_left=0, margin_right=0),
**(parser_options or {}),
)
renderer = renderer_factory(**(renderer_options or {}))
return run_pipeline(lines, parser=parser, renderer=renderer, base_path=base_path)

91
src/wapdown/models.py Normal file
View file

@ -0,0 +1,91 @@
from __future__ import annotations
from dataclasses import dataclass
from enum import Enum
from typing import List, Optional
class BlockKind(Enum):
PARAGRAPH = "paragraph"
HEADING = "heading"
CODE_BLOCK = "code_block"
BLOCKQUOTE = "blockquote"
LIST_ITEM = "list_item"
HORIZONTAL_RULE = "horizontal_rule"
TABLE = "table"
BLANK_LINE = "blank_line"
CARD_BREAK = "card_break"
@dataclass
class BlockStyle:
align: str = "left"
margin_left: int = 0
margin_right: int = 0
@dataclass
class StyleSpec:
align: Optional[str] = None
margin_left: Optional[int] = None
margin_right: Optional[int] = None
@dataclass
class ParagraphPayload:
text: str
@dataclass
class HeadingPayload:
level: int
text: str
@dataclass
class CodeBlockPayload:
lines: List[str]
@dataclass
class BlockQuotePayload:
depth: int
text: str
@dataclass
class ListItemPayload:
indent: str
marker: str
spacing: str
text: str
ordered: bool
@dataclass
class TablePayload:
rows: List[List[str]] # rows[0] is the header row
@dataclass
class CardBreakPayload:
"""Marks an explicit 'start a new card here' hint.
Emitted for the `{.card}` / `{.card Title}` directive, which is the
manual alternative to heading-level splitting.
"""
title: Optional[str] = None
@dataclass
class BlockEvent:
kind: BlockKind
payload: object
style: BlockStyle
stylable: bool = False
@dataclass
class StyleUpdateEvent:
spec: StyleSpec

View file

@ -0,0 +1,5 @@
"""Bundled parser implementations."""
from .markdown import MarkdownParser
__all__ = ["MarkdownParser"]

View file

@ -0,0 +1,589 @@
from __future__ import annotations
import json
import re
from typing import Iterable, Iterator, List, Optional, Union
from ..conversion.core import TABLE_SENTINEL_PREFIX
from ..models import (
BlockEvent,
BlockKind,
BlockQuotePayload,
BlockStyle,
CardBreakPayload,
CodeBlockPayload,
HeadingPayload,
ListItemPayload,
ParagraphPayload,
StyleSpec,
StyleUpdateEvent,
TablePayload,
)
HEADING_PATTERN = re.compile(r"^(#{1,6})\s+(.*)$")
ORDERED_LIST_PATTERN = re.compile(r"^(\s*)(\d+\.)(\s+)(.*)$")
UNORDERED_LIST_PATTERN = re.compile(r"^(\s*)([*+-])(\s+)(.*)$")
BLOCKQUOTE_PATTERN = re.compile(r"^\s{0,3}>(.*)$")
HORIZONTAL_RULE_PATTERN = re.compile(r"^\s*([-*_])(?:\s*\1){2,}\s*$")
CARD_BREAK_PATTERN = re.compile(r"^\s*\{\s*\.card(?:\s+(?P<title>.+?))?\s*\}\s*$")
INLINE_PARA_RE = re.compile(r"^\s*<p\b([^>]*)>(.*?)</p>\s*$", re.IGNORECASE)
PARA_OPEN_RE = re.compile(r"^\s*<p\b([^>]*)>\s*$", re.IGNORECASE)
PARA_CLOSE_RE = re.compile(r"^\s*</p>\s*$", re.IGNORECASE)
MMD_ATTR_LINE_RE = re.compile(r"^\{\s*:(.+)\}\s*$")
MMD_ATTR_TAIL_RE = re.compile(r"(.*?)\s*\{\s*:(.+?)\}\s*$")
class MarkdownParser:
def __init__(self, base_style: BlockStyle) -> None:
self._base_style = base_style
self._style_stack: List[BlockStyle] = [self._make_base_style()]
self._paragraph_style_spec: Optional[StyleSpec] = None
self._pending_block_style_spec: Optional[StyleSpec] = None
self._last_stylable_block: bool = False
def parse(self, lines: Iterable[str]) -> Iterator[Union[BlockEvent, StyleUpdateEvent]]:
self._reset_state()
iterator = iter(lines)
in_code_block = False
code_lines: List[str] = []
indented_code_lines: List[str] = []
current_paragraph: List[str] = []
for raw_line in iterator:
line = raw_line.rstrip("\n")
if line.startswith(TABLE_SENTINEL_PREFIX):
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
payload = self._build_table_payload(line)
style = self._combine_styles(self._current_style(), self._pending_block_style_spec)
self._pending_block_style_spec = None
self._last_stylable_block = True
yield BlockEvent(
kind=BlockKind.TABLE,
payload=payload,
style=style,
stylable=True,
)
continue
if in_code_block:
if line.strip().startswith("```"):
event = self._flush_code_block(code_lines)
if event is not None:
yield event
code_lines = []
in_code_block = False
else:
code_lines.append(line)
continue
if indented_code_lines:
if line.startswith(" "):
indented_code_lines.append(line[4:])
continue
if not line.strip():
event = self._flush_code_block(indented_code_lines)
if event is not None:
yield event
indented_code_lines = []
else:
event = self._flush_code_block(indented_code_lines)
if event is not None:
yield event
indented_code_lines = []
card_break_match = CARD_BREAK_PATTERN.match(line)
if card_break_match:
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
title = card_break_match.group("title")
self._last_stylable_block = False
yield BlockEvent(
kind=BlockKind.CARD_BREAK,
payload=CardBreakPayload(title=title.strip() if title else None),
style=self._clone_style(),
stylable=False,
)
continue
inline_para = INLINE_PARA_RE.match(line.strip())
if inline_para:
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
spec = self._style_spec_from_html_attributes(inline_para.group(1) or "")
self._push_style(spec)
content = inline_para.group(2)
if content:
current_paragraph.append(content)
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
self._pop_style()
continue
open_para = PARA_OPEN_RE.match(line)
if open_para:
spec = self._style_spec_from_html_attributes(open_para.group(1) or "")
self._push_style(spec)
continue
close_para = PARA_CLOSE_RE.match(line)
if close_para:
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
self._paragraph_style_spec = None
self._pop_style()
continue
stripped = line.strip()
attr_match = MMD_ATTR_LINE_RE.match(stripped)
if attr_match:
spec = self._parse_style_spec_from_tokens(attr_match.group(1))
if spec:
if current_paragraph:
self._paragraph_style_spec = self._merge_specs(self._paragraph_style_spec, spec)
elif self._last_stylable_block:
yield StyleUpdateEvent(spec)
else:
self._pending_block_style_spec = self._merge_specs(self._pending_block_style_spec, spec)
continue
if line.strip().startswith("```"):
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
in_code_block = True
code_lines = []
continue
if line.startswith(" "):
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
indented_code_lines = [line[4:]]
continue
heading_match = HEADING_PATTERN.match(line)
if heading_match:
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
level = len(heading_match.group(1))
heading_text = heading_match.group(2).strip()
heading_text, inline_spec = self._extract_trailing_attr(heading_text)
combined_spec = self._merge_specs(self._pending_block_style_spec, inline_spec)
style = self._combine_styles(self._current_style(), combined_spec)
self._pending_block_style_spec = None
self._last_stylable_block = True
yield BlockEvent(
kind=BlockKind.HEADING,
payload=HeadingPayload(level=level, text=heading_text),
style=style,
stylable=True,
)
continue
if HORIZONTAL_RULE_PATTERN.match(line):
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
self._last_stylable_block = False
yield BlockEvent(
kind=BlockKind.HORIZONTAL_RULE,
payload=None,
style=self._clone_style(),
stylable=False,
)
continue
if BLOCKQUOTE_PATTERN.match(line):
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
quote_event = self._parse_blockquote(line)
if quote_event is not None:
yield quote_event
continue
if UNORDERED_LIST_PATTERN.match(line) or ORDERED_LIST_PATTERN.match(line):
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
current_paragraph = []
list_event = self._parse_list_line(line)
if list_event is not None:
yield list_event
continue
if not line.strip():
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
yield BlockEvent(
kind=BlockKind.BLANK_LINE,
payload=None,
style=self._clone_style(),
stylable=False,
)
current_paragraph = []
continue
current_paragraph.append(line)
event = self._flush_paragraph(current_paragraph)
if event is not None:
yield event
if in_code_block:
final_event = self._flush_code_block(code_lines)
if final_event is not None:
yield final_event
if indented_code_lines:
final_event = self._flush_code_block(indented_code_lines)
if final_event is not None:
yield final_event
def _reset_state(self) -> None:
self._style_stack = [self._make_base_style()]
self._paragraph_style_spec = None
self._pending_block_style_spec = None
self._last_stylable_block = False
def _flush_paragraph(self, paragraph_lines: List[str]) -> Optional[BlockEvent]:
if not paragraph_lines:
return None
text = " ".join(line.strip() for line in paragraph_lines)
combined_spec = self._merge_specs(self._pending_block_style_spec, self._paragraph_style_spec)
style = self._combine_styles(self._current_style(), combined_spec)
paragraph_lines.clear()
self._paragraph_style_spec = None
self._pending_block_style_spec = None
self._last_stylable_block = True
return BlockEvent(
kind=BlockKind.PARAGRAPH,
payload=ParagraphPayload(text=text),
style=style,
stylable=True,
)
def _flush_code_block(self, code_lines: List[str]) -> Optional[BlockEvent]:
if not code_lines:
return None
lines = code_lines.copy()
code_lines.clear()
self._last_stylable_block = False
return BlockEvent(
kind=BlockKind.CODE_BLOCK,
payload=CodeBlockPayload(lines=lines),
style=self._clone_style(),
stylable=False,
)
def _parse_blockquote(self, line: str) -> Optional[BlockEvent]:
content = line
depth = 0
while content.lstrip().startswith(">"):
depth += 1
content = content.lstrip()[1:]
text = content.lstrip()
self._last_stylable_block = False
return BlockEvent(
kind=BlockKind.BLOCKQUOTE,
payload=BlockQuotePayload(depth=max(1, depth), text=text),
style=self._clone_style(),
stylable=False,
)
def _parse_list_line(self, line: str) -> Optional[BlockEvent]:
ordered = ORDERED_LIST_PATTERN.match(line)
unordered = UNORDERED_LIST_PATTERN.match(line)
if ordered:
indent, marker, spacing, rest = ordered.groups()
ordered_flag = True
elif unordered:
indent, marker, spacing, rest = unordered.groups()
ordered_flag = False
else:
return None
self._last_stylable_block = False
return BlockEvent(
kind=BlockKind.LIST_ITEM,
payload=ListItemPayload(
indent=indent,
marker=marker,
spacing=spacing,
text=rest,
ordered=ordered_flag,
),
style=self._clone_style(),
stylable=False,
)
def _build_table_payload(self, sentinel_line: str) -> TablePayload:
payload = sentinel_line[len(TABLE_SENTINEL_PREFIX) :]
try:
data = json.loads(payload)
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid table sentinel payload: {payload}") from exc
rows = data.get("rows", [])
return TablePayload(rows=[[str(cell) for cell in row] for row in rows])
def _make_base_style(self) -> BlockStyle:
return BlockStyle(
align=self._base_style.align,
margin_left=self._base_style.margin_left,
margin_right=self._base_style.margin_right,
)
def _current_style(self) -> BlockStyle:
return self._style_stack[-1]
def _clone_style(self) -> BlockStyle:
style = self._current_style()
return BlockStyle(
align=style.align,
margin_left=style.margin_left,
margin_right=style.margin_right,
)
def _push_style(self, spec: Optional[StyleSpec]) -> None:
base = self._current_style()
self._style_stack.append(self._combine_styles(base, spec))
def _pop_style(self) -> None:
if len(self._style_stack) > 1:
self._style_stack.pop()
def _combine_styles(self, base: BlockStyle, spec: Optional[StyleSpec]) -> BlockStyle:
if spec is None:
return BlockStyle(
align=base.align,
margin_left=base.margin_left,
margin_right=base.margin_right,
)
return BlockStyle(
align=spec.align or base.align,
margin_left=spec.margin_left if spec.margin_left is not None else base.margin_left,
margin_right=spec.margin_right if spec.margin_right is not None else base.margin_right,
)
def _merge_specs(self, first: Optional[StyleSpec], second: Optional[StyleSpec]) -> Optional[StyleSpec]:
if first is None and second is None:
return None
if first is None:
return second
if second is None:
return first
return StyleSpec(
align=second.align or first.align,
margin_left=second.margin_left if second.margin_left is not None else first.margin_left,
margin_right=second.margin_right if second.margin_right is not None else first.margin_right,
)
def _style_spec_from_html_attributes(self, attributes: str) -> Optional[StyleSpec]:
if not attributes:
return None
attr_pattern = re.compile(r"([\w:-]+)\s*=\s*(\".*?\"|'.*?'|\S+)")
attr_map = {name.lower(): value.strip().strip("\"'") for name, value in attr_pattern.findall(attributes)}
spec: Optional[StyleSpec] = None
align_value = attr_map.get("align")
if align_value:
normalized = self._normalize_align(align_value)
if normalized:
spec = self._merge_specs(spec, StyleSpec(align=normalized))
style_value = attr_map.get("style")
if style_value:
css_spec = self._style_spec_from_css(style_value)
spec = self._merge_specs(spec, css_spec)
return spec
def _style_spec_from_css(self, css: str) -> Optional[StyleSpec]:
spec = StyleSpec()
changed = False
for declaration in css.split(";"):
if ":" not in declaration:
continue
name, value = declaration.split(":", 1)
name = name.strip().lower()
value = value.strip()
if not value:
continue
if name == "text-align":
normalized = self._normalize_align(value)
if normalized:
spec.align = normalized
changed = True
elif name == "margin":
left, right, auto_center = self._parse_css_margin_shorthand(value)
if left is not None:
spec.margin_left = left
changed = True
if right is not None:
spec.margin_right = right
changed = True
if auto_center:
spec.align = "center"
changed = True
elif name == "margin-left":
parsed = self._parse_space_value(value)
if parsed is not None:
spec.margin_left = parsed
changed = True
elif value.lower() == "auto":
spec.align = spec.align or "center"
changed = True
elif name == "margin-right":
parsed = self._parse_space_value(value)
if parsed is not None:
spec.margin_right = parsed
changed = True
elif value.lower() == "auto":
spec.align = spec.align or "center"
changed = True
return spec if changed else None
def _parse_style_spec_from_tokens(self, token_str: str) -> Optional[StyleSpec]:
tokens = re.split(r"\s+", token_str.strip())
spec = StyleSpec()
changed = False
for token in tokens:
token = token.strip()
if not token:
continue
if token.startswith("."):
align = self._class_to_align(token[1:])
if align:
spec.align = align
changed = True
continue
if "=" in token:
key, value = token.split("=", 1)
key = key.strip().lower().lstrip(".")
value = value.strip().strip("\"'")
if key in {"align", "text-align"}:
normalized = self._normalize_align(value)
if normalized:
spec.align = normalized
changed = True
elif key in {"margin", "margin-left", "margin-right"}:
if key == "margin":
left, right, auto_center = self._parse_css_margin_shorthand(value)
if left is not None:
spec.margin_left = left
changed = True
if right is not None:
spec.margin_right = right
changed = True
if auto_center:
spec.align = "center"
changed = True
elif key == "margin-left":
parsed = self._parse_space_value(value)
if parsed is not None:
spec.margin_left = parsed
changed = True
elif value.lower() == "auto":
spec.align = spec.align or "center"
changed = True
elif key == "margin-right":
parsed = self._parse_space_value(value)
if parsed is not None:
spec.margin_right = parsed
changed = True
elif value.lower() == "auto":
spec.align = spec.align or "center"
changed = True
continue
align = self._normalize_align(token)
if align:
spec.align = align
changed = True
return spec if changed else None
def _parse_css_margin_shorthand(self, value: str):
parts = [part for part in re.split(r"\s+", value.strip()) if part]
if not parts:
return None, None, False
values: List[Optional[int]] = []
autos: List[bool] = []
for part in parts:
if part.lower() == "auto":
values.append(None)
autos.append(True)
else:
parsed = self._parse_space_value(part)
values.append(parsed)
autos.append(False)
left_auto = False
right_auto = False
if len(values) == 1:
left = right = values[0]
left_auto = right_auto = autos[0]
elif len(values) == 2:
left = right = values[1]
left_auto = right_auto = autos[1]
elif len(values) == 3:
left = right = values[1]
left_auto = right_auto = autos[1]
else:
right = values[1]
left = values[3]
right_auto = autos[1]
left_auto = autos[3]
auto_center = left_auto and right_auto
return left, right, auto_center
def _parse_space_value(self, value: str) -> Optional[int]:
match = re.match(r"(-?\d+(?:\.\d+)?)", value.strip())
if not match:
return None
number = float(match.group(1))
return max(0, int(round(number)))
def _normalize_align(self, value: str) -> Optional[str]:
normalized = value.strip().lower()
mapping = {
"centre": "center",
"center": "center",
"left": "left",
"right": "right",
}
return mapping.get(normalized)
def _class_to_align(self, class_name: str) -> Optional[str]:
name = class_name.strip().lower().lstrip(".")
if name in {"center", "text-center", "align-center"}:
return "center"
if name in {"left", "text-left", "align-left"}:
return "left"
if name in {"right", "text-right", "align-right"}:
return "right"
return None
def _extract_trailing_attr(self, text: str):
match = MMD_ATTR_TAIL_RE.match(text)
if not match:
return text, None
clean_text = match.group(1).rstrip()
spec = self._parse_style_spec_from_tokens(match.group(2))
return clean_text, spec

View file

@ -0,0 +1,44 @@
from __future__ import annotations
from typing import Any
from ..conversion.core import ParserFactory, RendererFactory
from .registry import PluginRegistry
parser_plugins = PluginRegistry[ParserFactory]()
renderer_plugins = PluginRegistry[RendererFactory[Any]]()
def register_parser(name: str, factory: ParserFactory) -> None:
parser_plugins.register(name, factory)
def register_renderer(name: str, factory: RendererFactory[Any]) -> None:
renderer_plugins.register(name, factory)
def get_parser_factory(name: str) -> ParserFactory:
return parser_plugins.get(name)
def get_renderer_factory(name: str) -> RendererFactory[Any]:
return renderer_plugins.get(name)
def available_parsers() -> list[str]:
return parser_plugins.names()
def available_renderers() -> list[str]:
return renderer_plugins.names()
__all__ = [
"available_parsers",
"available_renderers",
"get_parser_factory",
"get_renderer_factory",
"register_parser",
"register_renderer",
]

View file

@ -0,0 +1,25 @@
from __future__ import annotations
from typing import Dict, Generic, List, TypeVar
T = TypeVar("T")
class PluginRegistry(Generic[T]):
def __init__(self) -> None:
self._factories: Dict[str, T] = {}
def register(self, name: str, factory: T) -> None:
if name in self._factories:
raise ValueError(f"Plugin '{name}' is already registered.")
self._factories[name] = factory
def get(self, name: str) -> T:
try:
return self._factories[name]
except KeyError as exc:
raise KeyError(f"Plugin '{name}' is not registered.") from exc
def names(self) -> List[str]:
return sorted(self._factories.keys())

View file

@ -0,0 +1,5 @@
"""Bundled renderer implementations."""
from .wml import WMLRenderer
__all__ = ["WMLRenderer"]

View file

@ -0,0 +1,28 @@
"""WML deck renderer, registered under the name `wml`."""
from typing import Any
from ...plugins import register_renderer
from .deck import Card, Section
from .inline import escape_wml, process_inline
from .renderer import DeckOutput, WMLRenderer, WmlDocument
def _wml_renderer_factory(**options: Any) -> WMLRenderer:
return WMLRenderer(**options)
try:
register_renderer("wml", _wml_renderer_factory)
except ValueError: # pragma: no cover - already registered on re-import
pass
__all__ = [
"Card",
"DeckOutput",
"Section",
"WMLRenderer",
"WmlDocument",
"escape_wml",
"process_inline",
]

View file

@ -0,0 +1,194 @@
"""Document topology: how one Markdown file becomes a navigable deck.
This is the part that makes WML output unlike a plain text export. A text
renderer transforms a document; a deck *restructures* it into short,
button-navigated cards with invented navigation between them. Three
independent concerns live here:
1. Sectioning -- where card boundaries fall (`{.card}` or `split_level`).
2. Packing -- splitting an oversized section across physical cards.
3. Topology -- hub-and-spoke vs. linear, and the nav each card carries.
"""
from __future__ import annotations
import re
import sys
from dataclasses import dataclass, field
from typing import List, Optional, Tuple
from ...models import BlockEvent, BlockKind
# Headroom reserved per card for the nav markup a continuation card carries.
# When nav lives in a deck-level <template>, each card's own nav markup is
# effectively free and only the rare per-card override needs room.
NAV_HEADROOM_BYTES = 160
TEMPLATED_NAV_HEADROOM_BYTES = 24
@dataclass
class Section:
"""A logical run of the document destined for one card (before packing)."""
title: Optional[str]
events: List[BlockEvent]
@dataclass
class Card:
"""One physical <card>, after byte packing."""
id: str
title: Optional[str]
fragments: List[str] = field(default_factory=list)
back_label: Optional[str] = None
next_id: Optional[str] = None
next_label: Optional[str] = None
home_id: Optional[str] = None
deck_file: Optional[str] = None # set only in deck-per-card mode
def split_sections(
events: List[BlockEvent], split_level: int
) -> Tuple[List[BlockEvent], List[Section]]:
"""Return (preamble, sections).
Manual `{.card}` markers win over heading-level splitting when both are
present: an explicit boundary in the document is a stronger statement of
intent than a structural rule passed on the command line.
"""
card_breaks = [i for i, e in enumerate(events) if e.kind == BlockKind.CARD_BREAK]
if card_breaks:
return _split_by_card_breaks(events, card_breaks)
if split_level > 0:
headings = [
i
for i, e in enumerate(events)
if e.kind == BlockKind.HEADING and e.payload.level == split_level
]
if headings:
return _split_by_heading(events, headings)
return list(events), []
def _split_by_card_breaks(
events: List[BlockEvent], indices: List[int]
) -> Tuple[List[BlockEvent], List[Section]]:
preamble = events[: indices[0]]
boundaries = indices + [len(events)]
sections: List[Section] = []
for idx in range(len(indices)):
start = indices[idx] + 1 # skip the {.card} marker event itself
section_events = events[start : boundaries[idx + 1]]
marker_title = events[indices[idx]].payload.title
sections.append(
Section(
title=marker_title or first_heading_text(section_events),
events=section_events,
)
)
return preamble, sections
def _split_by_heading(
events: List[BlockEvent], indices: List[int]
) -> Tuple[List[BlockEvent], List[Section]]:
preamble = events[: indices[0]]
boundaries = indices + [len(events)]
sections: List[Section] = []
for idx in range(len(indices)):
start = indices[idx] # keep the heading event, it still renders in-body
section_events = events[start : boundaries[idx + 1]]
sections.append(
Section(title=events[indices[idx]].payload.text, events=section_events)
)
return preamble, sections
def first_heading_text(events: List[BlockEvent]) -> Optional[str]:
for event in events:
if event.kind == BlockKind.HEADING:
return event.payload.text
return None
def pack_section(
base_id: str,
title: Optional[str],
fragments: List[str],
*,
max_card_bytes: int,
headroom: int,
nav_next_label: str,
nav_prev_label: str,
entry_back_label: Optional[str],
exit_next_id: Optional[str],
exit_next_label: Optional[str],
home_id: Optional[str] = None,
) -> List[Card]:
"""Split one section's fragments across physical cards if they exceed
the byte budget. This is the overflow safety net, not the primary
splitting mechanism -- that is section-based splitting upstream.
"""
chunks = greedy_pack(fragments, base_id, max_card_bytes, headroom)
cards: List[Card] = []
last = len(chunks) - 1
for i, chunk in enumerate(chunks):
is_first = i == 0
is_last = i == last
next_id = f"{base_id}-p{i + 2}" if not is_last else exit_next_id
cards.append(
Card(
id=base_id if is_first else f"{base_id}-p{i + 1}",
title=title if is_first else None,
fragments=chunk,
back_label=entry_back_label if is_first else nav_prev_label,
next_id=next_id,
next_label=(nav_next_label if not is_last else exit_next_label),
home_id=home_id,
)
)
return cards
def greedy_pack(
fragments: List[str], base_id: str, max_card_bytes: int, headroom: int
) -> List[List[str]]:
if max_card_bytes <= 0:
return [fragments]
budget = max(1, max_card_bytes - headroom)
chunks: List[List[str]] = []
current: List[str] = []
current_size = 0
for frag in fragments:
frag_size = len(frag.encode("utf-8"))
if frag_size > budget:
if current:
chunks.append(current)
current, current_size = [], 0
chunks.append([frag])
print(
f"wapdown: warning: a single block in '{base_id}' is {frag_size} bytes, "
f"over the {budget}-byte budget (max_card_bytes {max_card_bytes} "
f"less {headroom} reserved for navigation); "
f"emitting it as its own card",
file=sys.stderr,
)
continue
if current and current_size + frag_size > budget:
chunks.append(current)
current, current_size = [], 0
current.append(frag)
current_size += frag_size
if current:
chunks.append(current)
return chunks or [[]]
_SLUG_STRIP_RE = re.compile(r"[^a-z0-9]+")
def slugify(text: Optional[str], fallback: str) -> str:
"""Filename-safe slug for deck-per-card output."""
slug = _SLUG_STRIP_RE.sub("-", (text or "").lower()).strip("-")
return slug[:40] or fallback

View file

@ -0,0 +1,125 @@
"""Inline Markdown -> WML markup mapping, plus WML's escaping rules."""
from __future__ import annotations
import re
import sys
from typing import List
CODE_STASH_RE = re.compile(r"`[^`]*`")
STRIKETHROUGH_RE = re.compile(r"~~(.*?)~~")
BOLD_RE = re.compile(r"\*\*(.*?)\*\*")
ITALIC_RE = re.compile(r"(?<!\*)\*(?!\*)(.*?)(?<!\*)\*(?!\*)")
UNDERLINE_STRONG_RE = re.compile(r"__(.*?)__")
UNDERLINE_EM_RE = re.compile(r"(?<!_)_(?!_)(.*?)(?<!_)_(?!_)")
IMAGE_RE = re.compile(r"!\[([^\]]*)\]\(([^)]+)\)")
LINK_RE = re.compile(r"(?<!\!)\[([^\]]+)\]\(([^)]+)\)")
_STASH = "\u0000{kind}{index}\u0000"
_CODE = "CODE"
_TAG = "TAG"
def escape_wml(text: str) -> str:
"""XML-escape text for WML element content or a double-quoted attribute
value, plus WML's own quirk: a literal '$' must become '$$' since WML
uses '$' for variable substitution. Applied once, up front, to the whole
raw string before any inline markup is turned into tags, so link/image
URLs captured out of the already-escaped text are correctly escaped for
attribute use too.
"""
text = text.replace("&", "&amp;")
text = text.replace("<", "&lt;").replace(">", "&gt;").replace('"', "&quot;")
return text.replace("$", "$$")
def process_inline(text: str, *, images: str = "keep") -> str:
"""Convert one run of inline Markdown into WML markup.
Links and images are resolved *before* the emphasis patterns run and
stashed behind sentinels, because a URL is not Markdown: a path like
`/a_b_c` or `/a*b*c` would otherwise be shredded into `<i>` tags by
UNDERLINE_EM_RE / ITALIC_RE before LINK_RE ever saw it.
"""
escaped = escape_wml(text)
code_segments: List[str] = []
tag_segments: List[str] = []
def stash_code(match: re.Match[str]) -> str:
code_segments.append(match.group(0)[1:-1])
return _STASH.format(kind=_CODE, index=len(code_segments) - 1)
def stash(markup: str) -> str:
tag_segments.append(markup)
return _STASH.format(kind=_TAG, index=len(tag_segments) - 1)
def stash_image(match: re.Match[str]) -> str:
alt = match.group(1).strip()
src = match.group(2).strip()
rendered = _render_image(alt, src, images)
return stash(rendered) if rendered else ""
def stash_link(match: re.Match[str]) -> str:
label = match.group(1).strip()
url = match.group(2).strip()
# WML 1.3 declares <a> as (#PCDATA | br | img)*, so a link label
# cannot carry emphasis at all -- there is nowhere valid to put a
# <b> inside it. The markers are stripped rather than translated.
return stash(f'<a href="{url}">{strip_emphasis(label)}</a>')
escaped = CODE_STASH_RE.sub(stash_code, escaped)
escaped = IMAGE_RE.sub(stash_image, escaped) # before LINK_RE: `![x](y)` is not a link
escaped = LINK_RE.sub(stash_link, escaped)
escaped = STRIKETHROUGH_RE.sub(lambda m: m.group(1), escaped) # no WML equivalent
escaped = BOLD_RE.sub(lambda m: f"<b>{m.group(1)}</b>", escaped)
escaped = UNDERLINE_STRONG_RE.sub(lambda m: f"<b>{m.group(1)}</b>", escaped)
escaped = ITALIC_RE.sub(lambda m: f"<i>{m.group(1)}</i>", escaped)
escaped = UNDERLINE_EM_RE.sub(lambda m: f"<i>{m.group(1)}</i>", escaped)
for index, markup in enumerate(tag_segments):
escaped = escaped.replace(_STASH.format(kind=_TAG, index=index), markup)
for index, code in enumerate(code_segments):
escaped = escaped.replace(_STASH.format(kind=_CODE, index=index), code)
return escaped
_EMPHASIS_MARKERS_RE = re.compile(r"\*\*|__|~~|[*_`]")
def strip_emphasis(text: str) -> str:
"""Drop inline emphasis markers, leaving their content.
Used where the WML content model admits no emphasis elements, such as a
link label or a <select> option.
"""
return _EMPHASIS_MARKERS_RE.sub("", text)
def _render_image(alt: str, src: str, policy: str) -> str:
"""Apply the image policy. No WBMP conversion is ever performed.
Most WAP 1.x browsers render only WBMP, so `alt` and `drop` exist to
degrade a document that references PNG/JPEG art rather than emit an
element the device will show as a broken-image placeholder.
"""
if policy == "keep" or _is_wbmp(src):
return f'<img src="{src}" alt="{alt}"/>'
if policy == "drop":
_warn_image(src, "dropped")
return ""
_warn_image(src, "replaced with its alt text")
return alt
def _is_wbmp(src: str) -> bool:
return src.split("?", 1)[0].split("#", 1)[0].lower().endswith(".wbmp")
def _warn_image(src: str, action: str) -> None:
print(
f"wapdown: warning: non-WBMP image '{src}' {action}; "
f"WAP 1.x browsers generally render only WBMP",
file=sys.stderr,
)

View file

@ -0,0 +1,434 @@
"""The WML deck renderer."""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Any, Dict, List, Optional
from ...config import DeckConfig
from ...models import BlockEvent, BlockKind, StyleUpdateEvent
from .deck import (
NAV_HEADROOM_BYTES,
TEMPLATED_NAV_HEADROOM_BYTES,
Card,
Section,
first_heading_text,
pack_section,
slugify,
split_sections,
)
from .inline import escape_wml, process_inline
WML_PROLOG = (
'<?xml version="1.0"?>\n'
'<!DOCTYPE wml PUBLIC "-//WAPFORUM//DTD WML 1.3//EN" '
'"http://www.wapforum.org/DTD/wml13.dtd">\n'
)
MENU_CARD_ID = "menu"
INDEX_DECK_FILE = "index.wml"
@dataclass
class WmlDocument:
"""One output file. `name` is None for single-deck output."""
text: str
name: Optional[str] = None
@dataclass
class DeckOutput:
documents: List[WmlDocument] = field(default_factory=list)
@property
def text(self) -> str:
"""The single deck's markup; only valid when one document was produced."""
if len(self.documents) != 1:
raise ValueError(
f"Deck produced {len(self.documents)} files; write them individually."
)
return self.documents[0].text
class WMLRenderer:
"""Render Markdown events as a WML deck for WAP browsers.
Unlike a text renderer, which can emit output incrementally per
`handle_event`, this buffers the whole event stream and does the real
work in `finalize()`: card boundaries, the menu card's choices, and
forward/backward nav targets all depend on the full document's
structure being known up front. It still satisfies the Renderer
protocol, so it plugs into `run_pipeline` unchanged.
"""
def __init__(self, config: Optional[DeckConfig] = None, **options: Any) -> None:
self.config = config or DeckConfig.resolve(overrides=options)
self._events: List[BlockEvent] = []
# base card id -> output filename; populated only in deck-per-card mode.
self._deck_files: Dict[str, str] = {}
# -- Renderer protocol ---------------------------------------------------
def handle_event(self, event: "BlockEvent | StyleUpdateEvent") -> None:
if isinstance(event, BlockEvent):
self._events.append(event)
# StyleUpdateEvent: WML has no per-block alignment/style stack to
# apply this to, by design; ignored like other structural mismatches.
def finalize(self) -> DeckOutput:
cfg = self.config
preamble, sections = split_sections(self._events, cfg.split_level)
multi_deck = cfg.deck_per_card and len(sections) >= 2
# Filenames must be known before any card is rendered, since the menu's
# choices and every cross-deck <go> need to name the target file.
self._deck_files = self._plan_deck_files(sections) if multi_deck else {}
if len(sections) < 2:
cards = self._build_single(preamble, sections)
elif cfg.menu:
cards = self._build_hub_and_spoke(preamble, sections)
else:
cards = self._build_linear(preamble, sections)
if multi_deck:
return self._emit_multi_deck(cards)
return DeckOutput([WmlDocument(text=self._emit_deck(cards))])
def _plan_deck_files(self, sections: List[Section]) -> Dict[str, str]:
planned: Dict[str, str] = {}
used: set[str] = {INDEX_DECK_FILE}
for i, section in enumerate(sections):
base = f"card{i + 1}"
planned[base] = self._unique(f"{slugify(section.title, base)}.wml", used)
return planned
# -- Card topology --------------------------------------------------------
def _build_single(
self, preamble: List[BlockEvent], sections: List[Section]
) -> List[Card]:
events = list(preamble)
if sections:
events.extend(sections[0].events)
# Only an explicit title becomes a card title here: the document's
# own first heading is already rendered into the card body, and a
# <card title=> would duplicate it on the device's screen heading.
return self._pack(
"card1",
self.config.title,
self._render_events(events),
entry_back_label=None,
exit_next_id=None,
exit_next_label=None,
)
def _build_hub_and_spoke(
self, preamble: List[BlockEvent], sections: List[Section]
) -> List[Card]:
cfg = self.config
menu_title = cfg.title or first_heading_text(preamble)
menu_fragments = self._render_events(preamble)
menu_fragments.extend(self._render_menu_choices(sections, menu_title))
# The menu card is packed like any other: a long preamble plus many
# choices is exactly the shape that overflows the byte budget.
cards = self._pack(
MENU_CARD_ID,
menu_title,
menu_fragments,
entry_back_label=None,
exit_next_id=None,
exit_next_label=None,
home_id=None,
)
for i, section in enumerate(sections):
cards.extend(
self._pack(
f"card{i + 1}",
section.title or "Untitled",
self._render_events(section.events),
entry_back_label=cfg.nav_back_label,
exit_next_id=None,
exit_next_label=None,
home_id=MENU_CARD_ID,
)
)
return cards
def _build_linear(
self, preamble: List[BlockEvent], sections: List[Section]
) -> List[Card]:
cfg = self.config
cards: List[Card] = []
total = len(sections)
for i, section in enumerate(sections):
is_first = i == 0
events = (list(preamble) + list(section.events)) if is_first else section.events
# Card 1 also carries the preamble, so prefer the deck-level
# title / preamble's own heading over this section's heading.
title = (
(cfg.title or first_heading_text(preamble) or section.title or "Untitled")
if is_first
else (section.title or "Untitled")
)
is_last = i == total - 1
next_id = f"card{i + 2}" if not is_last else None
cards.extend(
self._pack(
f"card{i + 1}",
title,
self._render_events(events),
entry_back_label=None if is_first else cfg.nav_back_label,
exit_next_id=next_id,
exit_next_label=cfg.nav_next_label if next_id else None,
)
)
return cards
def _pack(
self,
base_id: str,
title: Optional[str],
fragments: List[str],
*,
entry_back_label: Optional[str],
exit_next_id: Optional[str],
exit_next_label: Optional[str],
home_id: Optional[str] = None,
) -> List[Card]:
cfg = self.config
return pack_section(
base_id,
title,
fragments,
max_card_bytes=cfg.max_card_bytes,
headroom=self._headroom(),
nav_next_label=cfg.nav_next_label,
nav_prev_label=cfg.nav_prev_label,
entry_back_label=entry_back_label,
exit_next_id=exit_next_id,
exit_next_label=exit_next_label,
home_id=home_id if cfg.home_label else None,
)
def _headroom(self) -> int:
return TEMPLATED_NAV_HEADROOM_BYTES if self.config.template_nav else NAV_HEADROOM_BYTES
def _render_menu_choices(
self, sections: List[Section], menu_title: Optional[str]
) -> List[str]:
"""The hub's list of destinations.
`select` is the WAP-native form: a numeric keypad picks an option by
digit, where a list of anchors has to be scrolled through.
"""
targets = [
(escape_wml(section.title or "Untitled"), self._href(f"card{i + 1}"))
for i, section in enumerate(sections)
]
if self.config.menu_style == "select":
options = "".join(
f'<option onpick="{href}">{label}</option>' for label, href in targets
)
title = escape_wml(menu_title or "Menu")
return [f'<p><select title="{title}">{options}</select></p>']
return [f'<p><a href="{href}">{label}</a></p>' for label, href in targets]
# -- Deck emission --------------------------------------------------------
def _emit_deck(self, cards: List[Card], *, cards_in_file: Optional[List[Card]] = None) -> str:
body_cards = cards if cards_in_file is None else cards_in_file
parts = [WML_PROLOG, "<wml>\n"]
head = self._render_head()
if head:
parts.append(head + "\n")
template = self._render_template(cards)
if template:
parts.append(template + "\n")
parts.append("\n".join(self._render_card(card, template=bool(template)) for card in body_cards))
parts.append("\n</wml>\n")
return "".join(parts)
def _emit_multi_deck(self, cards: List[Card]) -> DeckOutput:
"""One file per section, cross-linked by filename.
This is the real answer to per-deck cache limits: `max_card_bytes`
can only paginate within a single file, which still has to be
fetched whole.
"""
self._assign_deck_files(cards)
by_file: Dict[str, List[Card]] = {}
for card in cards:
by_file.setdefault(card.deck_file or INDEX_DECK_FILE, []).append(card)
return DeckOutput(
[
WmlDocument(text=self._emit_deck(cards, cards_in_file=group), name=name)
for name, group in by_file.items()
]
)
def _assign_deck_files(self, cards: List[Card]) -> None:
for card in cards:
base = card.id.split("-p", 1)[0]
card.deck_file = (
INDEX_DECK_FILE if base == MENU_CARD_ID else self._deck_files.get(base, INDEX_DECK_FILE)
)
@staticmethod
def _unique(name: str, used: set[str]) -> str:
candidate, counter = name, 2
while candidate in used:
candidate = f"{name[:-4]}-{counter}.wml"
counter += 1
used.add(candidate)
return candidate
def _render_head(self) -> str:
cfg = self.config
entries: List[str] = []
if cfg.access_domain or cfg.access_path:
attrs = ""
if cfg.access_domain:
attrs += f' domain="{escape_wml(cfg.access_domain)}"'
if cfg.access_path:
attrs += f' path="{escape_wml(cfg.access_path)}"'
entries.append(f"<access{attrs}/>")
if cfg.cache_control is not None:
entries.append(
'<meta http-equiv="Cache-Control" '
f'content="max-age={cfg.cache_control}"/>'
)
return f"<head>\n{chr(10).join(entries)}\n</head>" if entries else ""
def _render_template(self, cards: List[Card]) -> str:
"""Hoist nav shared by every card into a deck-level <template>.
Only `<prev/>` and the Home softkey qualify: a forward `<go>` points
somewhere different from each card, so it cannot be shared. A card
that needs different nav than the template overrides it by declaring
a `<do>` of the same type (see `_render_card`).
"""
cfg = self.config
if not cfg.template_nav or len(cards) < 2:
return ""
entries: List[str] = []
if any(card.back_label for card in cards):
label = escape_wml(cfg.nav_back_label)
entries.append(f'<do type="prev" label="{label}"><prev/></do>')
if cfg.home_label and any(card.home_id for card in cards):
label = escape_wml(cfg.home_label)
href = self._href(MENU_CARD_ID, from_card=None)
entries.append(f'<do type="options" label="{label}"><go href="{href}"/></do>')
return f"<template>\n{chr(10).join(entries)}\n</template>" if entries else ""
def _render_card(self, card: Card, *, template: bool) -> str:
cfg = self.config
attrs = f' title="{escape_wml(card.title)}"' if card.title else ""
parts = [f'<card id="{card.id}"{attrs}>']
if template:
# Shadow the template's shared nav where this card differs from it.
# A <do> with the same type (its implicit name) replaces the
# inherited one; <noop/> suppresses it outright.
if not card.back_label:
parts.append('<do type="prev"><noop/></do>')
elif card.back_label != cfg.nav_back_label:
label = escape_wml(card.back_label)
parts.append(f'<do type="prev" label="{label}"><prev/></do>')
if cfg.home_label and not card.home_id:
parts.append('<do type="options"><noop/></do>')
else:
if card.back_label:
label = escape_wml(card.back_label)
parts.append(f'<do type="prev" label="{label}"><prev/></do>')
if cfg.home_label and card.home_id:
label = escape_wml(cfg.home_label)
href = self._href(card.home_id, from_card=card)
parts.append(f'<do type="options" label="{label}"><go href="{href}"/></do>')
if card.next_id:
label = escape_wml(card.next_label or cfg.nav_next_label)
href = self._href(card.next_id, from_card=card)
parts.append(f'<do type="accept" label="{label}"><go href="{href}"/></do>')
parts.extend(card.fragments)
parts.append("</card>")
return "\n".join(parts)
def _href(self, card_id: str, *, from_card: Optional[Card] = None) -> str:
"""Link to a card, crossing deck files when they exist."""
if not self.config.deck_per_card:
return f"#{card_id}"
base = card_id.split("-p", 1)[0]
target_file = (
INDEX_DECK_FILE if base == MENU_CARD_ID else self._deck_files.get(base, INDEX_DECK_FILE)
)
if from_card is not None and target_file == from_card.deck_file:
return f"#{card_id}"
return f"{target_file}#{card_id}"
# -- Block markup mapping -------------------------------------------------
def _render_events(self, events: List[BlockEvent]) -> List[str]:
fragments: List[str] = []
seen_heading = False
list_counter = 0
for event in events:
kind = event.kind
if kind == BlockKind.PARAGRAPH:
fragments.append(f"<p>{self._inline(event.payload.text)}</p>")
list_counter = 0
elif kind == BlockKind.HEADING:
text = self._inline(event.payload.text)
if not seen_heading:
fragments.append(f"<p><b><big>{text}</big></b></p>")
seen_heading = True
else:
fragments.append(f"<p><b>{text}</b></p>")
list_counter = 0
elif kind == BlockKind.LIST_ITEM:
text = self._inline(event.payload.text)
if event.payload.ordered:
list_counter += 1
fragments.append(f"<p>{list_counter}. {text}</p>")
else:
list_counter = 0
fragments.append(f"<p>- {text}</p>")
elif kind == BlockKind.CODE_BLOCK:
lines = [escape_wml(line) for line in event.payload.lines]
fragments.append(f"<p>{'<br/>'.join(lines)}</p>")
list_counter = 0
elif kind == BlockKind.BLOCKQUOTE:
fragments.append(f"<p><i>{self._inline(event.payload.text)}</i></p>")
list_counter = 0
elif kind == BlockKind.HORIZONTAL_RULE:
fragments.append("<p>------------</p>")
list_counter = 0
elif kind == BlockKind.TABLE:
fragments.append(self._render_table(event.payload.rows))
list_counter = 0
# BLANK_LINE, CARD_BREAK: no WML output of their own.
return fragments
def _render_table(self, rows: List[List[str]]) -> str:
if not rows:
return ""
columns = max(len(row) for row in rows)
tr_parts: List[str] = []
for row_index, row in enumerate(rows):
cells = list(row) + [""] * (columns - len(row))
td_parts = []
for cell in cells:
content = self._inline(cell)
if row_index == 0:
content = f"<b>{content}</b>"
td_parts.append(f"<td>{content}</td>")
tr_parts.append(f"<tr>{''.join(td_parts)}</tr>")
# <table> is flow content: the WML 1.3 content model for <card> is
# (onevent*, timer?, (do | p)*), so a bare <table> child is invalid.
return f'<p><table columns="{columns}">{"".join(tr_parts)}</table></p>'
def _inline(self, text: str) -> str:
return process_inline(text, images=self.config.images)