mews.page/md2mews.py

212 lines
7.7 KiB
Python
Raw Permalink Normal View History

2026-10-11 08:07:57 +03:00
#!/usr/bin/env python
# /// script
# requires-python = ">=3.10"
# dependencies = ["mistune>=3.0"]
# ///
"""Convert a Markdown file into a Mews page (Mews Profile 0.1).
Parsing is CommonMark (mistune) plus pipe tables; uv installs mistune
from the script header on the first run. The converter maps the tree onto
what mews-0.1.dtd permits: ATX and setext headings, paragraphs, emphasis,
strong, code, links, images (with titles), hard breaks, fenced and
indented code blocks, block quotes, ordered and unordered lists (with
start), and pipe tables. Raw HTML is escaped as text and table alignment
markers are dropped, because Mews has neither. Tables are recognized only
at the top level, matching the DTD, which allows them nowhere else.
Run with: uv run md2mews.py page.md -o page.html --lang en
"""
import argparse
import html
import sys
from pathlib import Path
import mistune
VERSION = "0.1"
DOCTYPE = (
'<!DOCTYPE html PUBLIC "-//WAPFORUM//DTD XHTML Mobile 1.2//EN"\n'
' "http://www.openmobilealliance.org/tech/DTD/xhtml-mobile12.dtd">'
)
VARIANTS = ("mews-warm", "mews-cool", "mews-green", "mews-mono")
MARKDOWN = mistune.create_markdown(renderer="ast", plugins=["table"])
def esc(text):
return html.escape(text)
def plain(nodes):
"""Text content of inline nodes, used for image alt text."""
parts = []
for node in nodes:
if node["type"] in ("text", "codespan", "inline_html"):
parts.append(node.get("raw", ""))
else:
parts.append(plain(node.get("children", [])))
return "".join(parts)
def inline_nodes(nodes):
out = []
for node in nodes:
kind, attrs = node["type"], node.get("attrs", {})
if kind == "text":
out.append(esc(node["raw"]))
elif kind == "codespan":
out.append(f"<code>{esc(node['raw'])}</code>")
elif kind == "emphasis":
out.append(f"<em>{inline_nodes(node['children'])}</em>")
elif kind == "strong":
out.append(f"<strong>{inline_nodes(node['children'])}</strong>")
elif kind == "link":
title = f' title="{esc(attrs["title"])}"' if attrs.get("title") else ""
out.append(f'<a href="{esc(attrs["url"])}"{title}>{inline_nodes(node["children"])}</a>')
elif kind == "image":
title = f' title="{esc(attrs["title"])}"' if attrs.get("title") else ""
alt = esc(plain(node["children"]))
out.append(f'<img src="{esc(attrs["url"])}" alt="{alt}"{title} />')
elif kind == "linebreak":
out.append("<br />")
elif kind == "softbreak":
out.append(" ")
elif kind == "inline_html":
out.append(esc(node["raw"]))
else:
sys.exit(f"error: unsupported inline markup: {kind}")
return "".join(out)
def render_block(node, level, no_table):
"""Render one block node into indented lines.
no_table is set inside block quotes and list items, where the Mews DTD
does not allow a table.
"""
pad = " " * level
kind, attrs = node["type"], node.get("attrs", {})
if kind in ("paragraph", "block_text"):
return [f"{pad}<p>{inline_nodes(node['children'])}</p>" if kind == "paragraph"
else f"{pad}{inline_nodes(node['children'])}"]
if kind == "heading":
tag = f"h{attrs['level']}"
return [f"{pad}<{tag}>{inline_nodes(node['children'])}</{tag}>"]
if kind == "block_quote":
return [f"{pad}<blockquote>", *render_blocks(node["children"], level + 1, True), f"{pad}</blockquote>"]
if kind == "block_code":
return [f"{pad}<pre>{esc(node['raw'].rstrip(chr(10)))}</pre>"]
if kind == "thematic_break":
return [f"{pad}<hr />"]
if kind == "block_html":
return [f"{pad}<p>{esc(node['raw'].strip())}</p>"]
if kind == "list":
tag = "ol" if attrs["ordered"] else "ul"
start = f' start="{attrs["start"]}"' if attrs.get("start") not in (None, 1) else ""
out = [f"{pad}<{tag}{start}>"]
for item in node["children"]:
out.extend(render_item(item, level + 1))
out.append(f"{pad}</{tag}>")
return out
if kind == "table":
if no_table:
sys.exit("error: a table inside a quote or list item cannot be a Mews page")
return render_table(node, pad)
sys.exit(f"error: unsupported Markdown block: {kind}")
def render_item(node, level):
pad = " " * level
children = node["children"]
if len(children) == 1 and children[0]["type"] == "block_text":
return [f"{pad}<li>{inline_nodes(children[0]['children'])}</li>"]
out = [f"{pad}<li>", *render_blocks(children, level + 1, True), f"{pad}</li>"]
return out
def render_blocks(nodes, level, no_table):
lines = []
for node in nodes:
if node["type"] == "blank_line":
continue
lines.extend(render_block(node, level, no_table))
return lines
def render_table(node, pad):
out = [f"{pad}<table>"]
for section in node["children"]:
rows = [section] if section["type"] == "table_head" else section["children"]
for row in rows:
cells = ""
for cell in row["children"]:
tag = "th" if cell["attrs"].get("head") else "td"
cells += f"<{tag}>{inline_nodes(cell['children'])}</{tag}>"
out.append(f"{pad} <tr>{cells}</tr>")
out.append(f"{pad}</table>")
return out
def page(title, body, language, direction, variant, description):
attrs = f'xml:lang="{language}" lang="{language}"'
if direction:
attrs += f' dir="{direction}"'
head = [
f" <title>{esc(title)}</title>",
f' <meta name="mews-profile" content="{VERSION}" />',
' <meta name="viewport" content="width=device-width" />',
]
if description:
head.append(f' <meta name="description" content="{esc(description)}" />')
head.append(' <link rel="stylesheet" type="text/css" href="mews-0.1.css" />')
body_attrs = f' class="{variant}"' if variant else ""
return "\n".join(
[
'<?xml version="1.0" encoding="UTF-8"?>',
DOCTYPE,
f'<html xmlns="http://www.w3.org/1999/xhtml" {attrs}>',
" <head>",
*head,
" </head>",
f" <body{body_attrs}>",
*body,
" </body>",
"</html>",
"",
]
)
def main():
parser = argparse.ArgumentParser(description="Convert a Markdown file into a Mews page.")
parser.add_argument("source", help="Markdown file to convert")
parser.add_argument("-o", "--output", help='output file, or "-" for stdout (default: source name with .html)')
parser.add_argument("--title", help="page title (default: first top-level heading)")
parser.add_argument("--description", help="meta description")
parser.add_argument("--lang", default="en", help="page language (default: en)")
parser.add_argument("--dir", choices=("ltr", "rtl"), help="text direction")
parser.add_argument("--variant", choices=VARIANTS, help="body variant class (SPEC.md 5.3)")
args = parser.parse_args()
source = Path(args.source)
blocks = MARKDOWN(source.read_text(encoding="utf-8-sig"))
title = args.title
if not title:
title = next(
(plain(block["children"]) for block in blocks if block["type"] == "heading" and block["attrs"]["level"] == 1),
None,
) or source.stem
result = page(title, render_blocks(blocks, 1, False), args.lang, args.dir, args.variant, args.description)
if args.output == "-":
sys.stdout.write(result)
else:
output = Path(args.output) if args.output else source.with_suffix(".html")
output.write_text(result, encoding="utf-8")
print(f"wrote {output} ({title})")
if __name__ == "__main__":
main()