md2amb tool

This commit is contained in:
randogoth 2025-10-20 16:59:52 +03:00
parent 0b11668f39
commit 4b58e21dfc
5 changed files with 355 additions and 37 deletions

View file

@ -5,7 +5,7 @@ This repository contains command line helpers for transforming Markdown into for
## Tools ## Tools
- `md2amb.py` converts Markdown into Amber-screen formatted text (see script for details). - `md2amb.py` converts Markdown into Amber-screen formatted text (see script for details).
- `md2txt.py` converts Markdown into 80-column, DOS-compatible plain text with extensive formatting support. It ships with the default `markdown` parser and `text` renderer plugins, registers optional `micron` and `amb` renderers for Micron/Ancient Machine Book output, and exposes the core pipeline so you can add your own parser or renderer modules: - `md2txt.py` converts Markdown into 80-column, DOS-compatible plain text with extensive formatting support. It ships with the default `markdown` parser and `text` renderer plugins, registers optional `micron` and `ama` renderers for Micron/Ancient Machine Book output, and exposes the core pipeline so you can add your own parser or renderer modules:
- FIGlet-rendered headings (H1H3) driven by optional YAML frontmatter (`h1_font`, `h2_font`, `h3_font`). - FIGlet-rendered headings (H1H3) driven by optional YAML frontmatter (`h1_font`, `h2_font`, `h3_font`).
- H4+ headings rendered in uppercase with dashed underlines. - H4+ headings rendered in uppercase with dashed underlines.
- Emphasis styles converted to spaced or delimited characters, e.g. `**bold**``B O L D`, `__strong__``_s_t_r_o_n_g_`, `~~strike~~``~s~t~r~i~k~e~`. - Emphasis styles converted to spaced or delimited characters, e.g. `**bold**``B O L D`, `__strong__``_s_t_r_o_n_g_`, `~~strike~~``~s~t~r~i~k~e~`.
@ -30,10 +30,16 @@ python md2txt.py input.md -o output.txt # convert to DOS
python md2txt.py input.md # write result to stdout python md2txt.py input.md # write result to stdout
python md2txt.py input.md --width 72 # override column width python md2txt.py input.md --width 72 # override column width
python md2txt.py input.md --parser markdown --renderer micron # emit Micron-formatted output python md2txt.py input.md --parser markdown --renderer micron # emit Micron-formatted output
python md2txt.py input.md --renderer amb # emit AMB/AMA markup python md2txt.py input.md --renderer ama # emit AMB/AMA markup
python md2txt.py input.md --renderer-option width=68 # pass KEY=VALUE to a renderer python md2txt.py input.md --renderer-option width=68 # pass KEY=VALUE to a renderer
``` ```
- `md2amb.py` package Markdown (and linked Markdown files) into a self-contained `.amb` archive composed of `.ama` articles that honour the 78-column/64 KiB AMA constraints.
```bash
python md2amb.py --title "Your Manual" docs/index.md output/manual.amb
```
`--parser` and `--renderer` select a plugin by name (defaults are `markdown` and `text`). Repeatable `--parser-option KEY=VALUE` and `--renderer-option KEY=VALUE` pairs are forwarded to the plugin factories as keyword arguments in addition to the defaults supplied by the CLI. Both scripts accept `--help` for the full option list. `--parser` and `--renderer` select a plugin by name (defaults are `markdown` and `text`). Repeatable `--parser-option KEY=VALUE` and `--renderer-option KEY=VALUE` pairs are forwarded to the plugin factories as keyword arguments in addition to the defaults supplied by the CLI. Both scripts accept `--help` for the full option list.
## FIGlet Fonts via Frontmatter ## FIGlet Fonts via Frontmatter

View file

@ -1,5 +1,6 @@
from __future__ import annotations from __future__ import annotations
import re
from functools import partial from functools import partial
from pathlib import Path from pathlib import Path
from typing import Any, Callable, List from typing import Any, Callable, List
@ -36,22 +37,14 @@ class AmaRenderer(TextRenderer):
super().__init__(width=min(width, 78), frontmatter=frontmatter) super().__init__(width=min(width, 78), frontmatter=frontmatter)
self.links.clear() self.links.clear()
self.link_indices.clear() self.link_indices.clear()
self._base_style = BlockStyle(align="left", margin_left=0, margin_right=0)
# Block rendering ----------------------------------------------------- # Block rendering -----------------------------------------------------
def _render_paragraph(self, payload: ParagraphPayload, style: BlockStyle) -> None: def _render_paragraph(self, payload: ParagraphPayload, style: BlockStyle) -> None:
processed = self._process_inline(payload.text) processed = self._process_inline(payload.text)
self._emit_render( def render_fn(target_style: BlockStyle) -> List[str]:
lambda target_style: self._wrap_text( return self._wrap_and_format(processed, target_style)
processed,
initial_indent="", self._emit_block(render_fn(style), stylable=True, render_fn=render_fn, style=style)
subsequent_indent="",
style=target_style,
hyphenate=self.hyphenate,
),
style,
stylable=True,
)
if self.paragraph_spacing > 0: if self.paragraph_spacing > 0:
self.output.extend([""] * self.paragraph_spacing) self.output.extend([""] * self.paragraph_spacing)
@ -84,17 +77,10 @@ class AmaRenderer(TextRenderer):
def _render_blockquote(self, payload: BlockQuotePayload, style: BlockStyle) -> None: def _render_blockquote(self, payload: BlockQuotePayload, style: BlockStyle) -> None:
processed = self._process_inline(payload.text) processed = self._process_inline(payload.text)
indent = " " * (3 * max(1, payload.depth)) indent = " " * (3 * max(1, payload.depth))
self._emit_render( def render_fn(target_style: BlockStyle) -> List[str]:
lambda target_style: self._wrap_text( return self._wrap_and_format(processed, target_style, initial_indent=indent, subsequent_indent=indent)
processed,
initial_indent=indent, self._emit_block(render_fn(style), stylable=True, render_fn=render_fn, style=style)
subsequent_indent=indent,
style=target_style,
hyphenate=self.hyphenate,
),
style,
stylable=True,
)
def _render_list_item(self, payload: ListItemPayload, style: BlockStyle) -> None: def _render_list_item(self, payload: ListItemPayload, style: BlockStyle) -> None:
base_indent = payload.indent.replace("\t", " ") base_indent = payload.indent.replace("\t", " ")
@ -104,17 +90,15 @@ class AmaRenderer(TextRenderer):
initial = f"{base_indent}{marker_indent}{marker}{spacing}" initial = f"{base_indent}{marker_indent}{marker}{spacing}"
subsequent = f"{base_indent}{marker_indent}{' ' * len(marker)}{spacing}" subsequent = f"{base_indent}{marker_indent}{' ' * len(marker)}{spacing}"
processed = self._process_inline(payload.text) processed = self._process_inline(payload.text)
self._emit_render( def render_fn(target_style: BlockStyle) -> List[str]:
lambda target_style: self._wrap_text( return self._wrap_and_format(
processed, processed,
target_style,
initial_indent=initial, initial_indent=initial,
subsequent_indent=subsequent, subsequent_indent=subsequent,
style=target_style, )
hyphenate=self.hyphenate,
), self._emit_block(render_fn(style), stylable=True, render_fn=render_fn, style=style)
style,
stylable=True,
)
def _render_horizontal_rule(self, _payload: object, style: BlockStyle) -> None: def _render_horizontal_rule(self, _payload: object, style: BlockStyle) -> None:
margin_left, _, available = self._margins(style) margin_left, _, available = self._margins(style)
@ -179,10 +163,6 @@ class AmaRenderer(TextRenderer):
return f"{alt} ({formatted_url})" return f"{alt} ({formatted_url})"
return formatted_url return formatted_url
def _combine_styles(self, base: BlockStyle, spec: StyleSpec | None) -> BlockStyle:
combined = super()._combine_styles(base, spec)
return BlockStyle(align=combined.align, margin_left=0, margin_right=0)
def _emphasis_handler(self, prefix: str, transform: Callable[[str], str]) -> Callable[[Any], str]: def _emphasis_handler(self, prefix: str, transform: Callable[[str], str]) -> Callable[[Any], str]:
def handler(match) -> str: def handler(match) -> str:
content = transform(match.group(1)) content = transform(match.group(1))
@ -190,6 +170,56 @@ class AmaRenderer(TextRenderer):
return handler return handler
def _wrap_and_format(
self,
text: str,
style: BlockStyle,
*,
initial_indent: str = "",
subsequent_indent: str | None = None,
) -> List[str]:
lines = self._wrap_text(
text,
initial_indent=initial_indent,
subsequent_indent=subsequent_indent if subsequent_indent is not None else initial_indent,
style=style,
hyphenate=self.hyphenate,
)
return self._propagate_modes(lines)
def _propagate_modes(self, lines: List[str]) -> List[str]:
mode = "%t"
result: List[str] = []
for line in lines:
current_line = line
if mode in {"%!", "%b"}:
stripped = current_line.lstrip()
leading_ws = current_line[: len(current_line) - len(stripped)]
while stripped.startswith(("%t", "%!", "%b")):
stripped = stripped[2:]
spacer = " " if stripped and not stripped.startswith(" ") else ""
current_line = f"{leading_ws}{mode}{spacer}{stripped}" if stripped else f"{leading_ws}{mode}"
end_mode = self._line_end_mode(current_line)
result.append(current_line)
mode = end_mode if end_mode in {"%!", "%b"} else "%t"
return result
@staticmethod
def _line_end_mode(line: str) -> str:
mode = "%t"
idx = 0
while True:
pos = line.find("%", idx)
if pos == -1 or pos + 1 >= len(line):
break
code = line[pos:pos + 2]
if code in {"%!", "%b", "%t"}:
mode = code
elif code == "%%":
pass
idx = pos + 2
return mode
def _ama_renderer_factory(*, frontmatter: FrontMatter, **options: Any) -> AmaRenderer: def _ama_renderer_factory(*, frontmatter: FrontMatter, **options: Any) -> AmaRenderer:
width = int(options.get("width", 78)) width = int(options.get("width", 78))

BIN
lorem.amb Normal file

Binary file not shown.

282
md2amb.py
View file

@ -0,0 +1,282 @@
#!/usr/bin/env python3
from __future__ import annotations
import argparse
import re
import struct
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, Iterable, List, Tuple
import ama_renderer # noqa: F401 - ensure AMA renderer plugin registration
from conversion_core import parse_frontmatter, run_conversion
from markdown_parser import MarkdownParser
from md_types import BlockStyle, FrontMatter
from plugins import get_parser_factory, get_renderer_factory, register_parser
from text_renderer import TextRenderer
# Ensure markdown parser registered for standalone usage
def _markdown_parser_factory(*, base_style: BlockStyle, **_: object) -> MarkdownParser:
return MarkdownParser(base_style)
try:
register_parser("markdown", _markdown_parser_factory)
except ValueError:
pass
MARKDOWN_LINK_RE = re.compile(r"(\[[^\]]*\]\()([^)]+)(\))")
LOCAL_LINK_RE = re.compile(r"^[A-Za-z0-9_.~/\\-]+$")
EXT_MD = {".md", ".markdown", ".mkd", ".mkdn"}
AMA_MAX_BYTES = 65_535
AMB_MAGIC = b"AMB1"
LINK_CONTINUE_LABEL = "Continue"
@dataclass
class Article:
source: Path
ama_name: str
def main(argv: Iterable[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="Convert Markdown into an AMB archive.")
parser.add_argument("input", type=Path, help="Root Markdown file to convert.")
parser.add_argument("output", type=Path, help="Output AMB filename.")
parser.add_argument("--title", type=str, help="Optional book title.")
args = parser.parse_args(list(argv) if argv is not None else None)
input_path = args.input.resolve()
if not input_path.exists():
parser.error(f"Input file '{input_path}' does not exist.")
amb_bytes = build_amb(
root_markdown=input_path,
title=args.title,
)
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_bytes(amb_bytes)
print(str(args.output))
return 0
def build_amb(root_markdown: Path, title: str | None) -> bytes:
articles = collect_articles(root_markdown)
ama_contents = render_articles(articles)
files = assemble_files(ama_contents, title)
return pack_amb(files)
def collect_articles(root_markdown: Path) -> Dict[Path, Article]:
queue: deque[Path] = deque([root_markdown])
visited: Dict[Path, Article] = {}
assigned_names: set[str] = set()
while queue:
current = queue.popleft()
current = current.resolve()
if current in visited:
continue
if not current.exists():
raise FileNotFoundError(f"Referenced file '{current}' was not found.")
if current == root_markdown:
ama_name = "INDEX.AMA"
else:
ama_name = assign_ama_name(current.stem, assigned_names)
assigned_names.add(ama_name)
visited[current] = Article(source=current, ama_name=ama_name)
for linked in find_local_markdown_links(current):
queue.append(linked)
return visited
def find_local_markdown_links(markdown_path: Path) -> List[Path]:
text = markdown_path.read_text(encoding="utf-8")
results: List[Path] = []
for _, target, _ in MARKDOWN_LINK_RE.findall(text):
cleaned = target.strip()
if not cleaned or cleaned.startswith("#"):
continue
if "://" in cleaned or cleaned.startswith(("mailto:", "ftp:", "gopher:", "tel:")):
continue
resolved = (markdown_path.parent / cleaned.split("#", 1)[0]).resolve()
if resolved.suffix.lower() in EXT_MD:
results.append(resolved)
return results
def assign_ama_name(stem: str, existing: set[str]) -> str:
base = "".join((c if c.isalnum() else "_") for c in stem.upper())
if not base:
base = "ARTICLE"
if base[0].isdigit():
base = f"_{base}"
base = base[:8]
name = f"{base}.AMA"
counter = 1
while name in existing:
suffix = f"{counter:02d}"
trimmed = base[: max(1, 8 - len(suffix))]
name = f"{trimmed}{suffix}.AMA"
counter += 1
return name
def render_articles(articles: Dict[Path, Article]) -> Dict[str, List[str]]:
parser_factory = get_parser_factory("markdown")
renderer_factory = get_renderer_factory("ama")
rendered: Dict[str, List[str]] = {}
for path, article in articles.items():
content = path.read_text(encoding="utf-8")
rewritten = rewrite_links(content, path.parent, articles)
frontmatter, body_lines = parse_frontmatter(rewritten.splitlines(keepends=True))
ama_lines = run_conversion(
body_lines,
frontmatter=frontmatter,
parser_factory=parser_factory,
renderer_factory=renderer_factory,
renderer_options={"width": 78},
base_path=path.parent,
)
split_articles = split_article(article.ama_name, ama_lines)
rendered.update(split_articles)
return rendered
def rewrite_links(markdown: str, base_dir: Path, articles: Dict[Path, Article]) -> str:
def replacer(match: re.Match[str]) -> str:
prefix, target, suffix = match.groups()
cleaned = target.strip()
candidate = (base_dir / cleaned.split("#", 1)[0]).resolve()
if candidate in articles:
mapped = articles[candidate].ama_name
return f"{prefix}{mapped}{suffix}"
return match.group(0)
return MARKDOWN_LINK_RE.sub(replacer, markdown)
def split_article(filename: str, lines: List[str]) -> Dict[str, List[str]]:
encoded = "\n".join(lines).encode("utf-8")
if len(encoded) <= AMA_MAX_BYTES:
return {filename: lines}
segments: List[List[str]] = []
current: List[str] = []
current_size = 0
def flush_segment() -> None:
nonlocal current, current_size
if current:
segments.append(current)
current = []
current_size = 0
for line in lines:
candidate_size = current_size + len((line + "\n").encode("utf-8"))
if candidate_size > AMA_MAX_BYTES and current:
flush_segment()
current.append(line)
current_size += len((line + "\n").encode("utf-8"))
flush_segment()
result: Dict[str, List[str]] = {}
stem = Path(filename).stem
generated_names = [filename]
for idx in range(1, len(segments)):
suffix = f"{idx:02d}"
trimmed = stem[: max(1, 8 - len(suffix))]
new_name = f"{trimmed}{suffix}.AMA"
counter = 1
while new_name in result or new_name in generated_names:
suffix = f"{idx:02d}{counter}"
trimmed = stem[: max(1, 8 - len(suffix))]
new_name = f"{trimmed}{suffix}.AMA"
counter += 1
generated_names.append(new_name)
for name, segment in zip(generated_names, segments, strict=False):
result[name] = segment[:]
for idx, name in enumerate(generated_names[:-1]):
next_name = generated_names[idx + 1]
result[name].append("")
result[name].append(f"%l{next_name}:{LINK_CONTINUE_LABEL}%t")
return result
def assemble_files(ama_contents: Dict[str, List[str]], title: str | None) -> List[Tuple[str, bytes]]:
files: List[Tuple[str, bytes]] = []
if title:
files.append(("TITLE", title.encode("ascii", "ignore")[:64]))
index_bytes = encode_ama("INDEX.AMA", ama_contents.pop("INDEX.AMA"))
files.append(("INDEX.AMA", index_bytes))
for name, lines in sorted(ama_contents.items()):
files.append((name, encode_ama(name, lines)))
return files
def encode_ama(name: str, lines: List[str]) -> bytes:
content = "\n".join(lines).rstrip("\n") + "\n"
data = content.encode("utf-8")
if len(data) > AMA_MAX_BYTES:
raise ValueError(f"Generated AMA article '{name}' exceeds {AMA_MAX_BYTES} bytes.")
if any("\t" in line for line in lines):
raise ValueError(f"Generated AMA article '{name}' contains tab characters.")
return data
def pack_amb(files: List[Tuple[str, bytes]]) -> bytes:
entries = []
offset = 6 + 20 * len(files)
payloads = []
for filename, data in files:
canonical = filename.upper()
if len(canonical) > 12:
raise ValueError(f"Filename '{canonical}' does not fit 8.3 constraints.")
payloads.append(data)
checksum = bsd_checksum(data)
entries.append((canonical, offset, len(data), checksum))
offset += len(data)
output = bytearray()
output.extend(AMB_MAGIC)
output.extend(struct.pack("<H", len(entries)))
for name, file_offset, length, checksum in entries:
padded = name.encode("ascii", "ignore")
padded = padded + b"\x00" * (12 - len(padded))
output.extend(padded)
output.extend(struct.pack("<I", file_offset))
output.extend(struct.pack("<H", length))
output.extend(struct.pack("<H", checksum))
for data in payloads:
output.extend(data)
return bytes(output)
def bsd_checksum(data: bytes) -> int:
checksum = 0
for byte in data:
checksum = (checksum >> 1) | ((checksum & 1) << 15)
checksum = (checksum + byte) & 0xFFFF
return checksum
if __name__ == "__main__":
raise SystemExit(main())

BIN
output.amb Normal file

Binary file not shown.