From 2db4eab94cd11372e30bdeb36524ee1f62bc24e3 Mon Sep 17 00:00:00 2001 From: randogoth Date: Mon, 20 Oct 2025 21:43:08 +0300 Subject: [PATCH] codepage --- README.md | 4 +- mambler.py | 431 ++++++++++++++++++++++++++++++++++++++++------------- 2 files changed, 328 insertions(+), 107 deletions(-) diff --git a/README.md b/README.md index 56a3e91..6ee47f6 100644 --- a/README.md +++ b/README.md @@ -12,11 +12,13 @@ The project wraps the [md2txt](../md2txt) toolchain, using its Markdown parser a ### Usage ```bash -uv run mambler.py --title "Your Book Title" path/to/index.md output.amb +uv run mambler.py --title "Your Book Title" --codepage 437 path/to/index.md output.amb ``` - `index.md` is the root Markdown file. Any local Markdown links it contains will be followed and bundled automatically. - `--title` is optional; when provided the value is embedded in the AMB archive header (truncated to 64 ASCII bytes). +- `--codepage` controls the 8-bit encoding used for every AMA article (default: `437`). Any character that cannot be expressed in the chosen codepage aborts the build with a helpful error so you can pick a better fit. +- If any emitted byte lives in the 0x80–0xFF range, `mambler` automatically writes a companion `UNICODE.MAP` file describing the high-half character mapping, mirroring the recommendation in the AMA/AMB specification. - The command prints the path of the generated AMB file on success. ### Development Notes diff --git a/mambler.py b/mambler.py index 0c55843..e2b4cc9 100644 --- a/mambler.py +++ b/mambler.py @@ -2,12 +2,13 @@ from __future__ import annotations import argparse +import codecs import re import struct from collections import deque from dataclasses import dataclass from pathlib import Path -from typing import Dict, Iterable, List, Tuple +from typing import Callable, Dict, Iterable, List, Tuple from md2txt import convert_markdown from md2txt.conversion.core import parse_frontmatter @@ -18,7 +19,234 @@ EXT_MD = {".md", ".markdown", ".mkd", ".mkdn"} AMA_MAX_BYTES = 65_535 AMB_MAGIC = b"AMB1" LINK_CONTINUE_LABEL = "Continue" -CONTINUE_OVERHEAD = len("\n".encode("utf-8")) + len((f"%l{'ABCDEFGH.AMA'}:{LINK_CONTINUE_LABEL}%t\n").encode("utf-8")) + + +@dataclass(frozen=True) +class CodepageInfo: + canonical: str + encoder: Callable[[str], bytes] + unicode_map: Tuple[int, ...] + + def encode(self, text: str) -> bytes: + return self.encoder(text) + + +CODEPAGE_ALIASES: Dict[str, str] = { + "cp437": "cp437", + "ibm437": "cp437", + "dos437": "cp437", + "437": "cp437", + "cp775": "cp775", + "775": "cp775", + "cp808": "cp808", + "808": "cp808", + "cp850": "cp850", + "850": "cp850", + "cp852": "cp852", + "852": "cp852", + "cp857": "cp857", + "857": "cp857", + "cp858": "cp858", + "858": "cp858", + "cp866": "cp866", + "866": "cp866", + "cp1250": "cp1250", + "1250": "cp1250", + "windows1250": "cp1250", + "win1250": "cp1250", + "cp1252": "cp1252", + "1252": "cp1252", + "windows1252": "cp1252", + "win1252": "cp1252", + "kam": "kam", + "kamenicky": "kam", + "kamenickyencoding": "kam", + "maz": "maz", + "mazovia": "maz", +} + +CODEPAGE_CACHE: Dict[str, CodepageInfo] = {} + + +def resolve_codepage(name: str) -> CodepageInfo: + normalized = _normalize_codepage_name(name) + try: + return CODEPAGE_CACHE[normalized] + except KeyError: + info = _build_codepage(normalized) + CODEPAGE_CACHE[normalized] = info + return info + + +def _normalize_codepage_name(raw: str) -> str: + token = raw.strip().lower() + token = token.replace("-", "").replace("_", "") + if token in CODEPAGE_ALIASES: + return CODEPAGE_ALIASES[token] + if token.startswith("ibm") and token[3:].isdigit(): + return f"cp{token[3:]}" + if token.startswith("dos") and token[3:].isdigit(): + return f"cp{token[3:]}" + if token.startswith("windows") and token[7:].isdigit(): + return f"cp{token[7:]}" + if token.startswith("win") and token[3:].isdigit(): + return f"cp{token[3:]}" + if token.isdigit(): + return f"cp{token}" + return token + + +def _build_codepage(canonical: str) -> CodepageInfo: + if canonical == "cp808": + return _build_cp808() + if canonical == "kam": + return _build_kam() + if canonical == "maz": + return _build_maz() + try: + codec_info = codecs.lookup(canonical) + except LookupError as exc: + raise ValueError(f"Unsupported codepage '{canonical}'.") from exc + + def encoder(text: str) -> bytes: + return text.encode(codec_info.name, "strict") + + try: + high_bytes = bytes(range(128, 256)).decode(codec_info.name, "strict") + except UnicodeDecodeError as exc: + raise ValueError(f"Codepage '{canonical}' is not an 8-bit single-byte encoding.") from exc + unicode_map = tuple(ord(ch) for ch in high_bytes) + return CodepageInfo(canonical=codec_info.name, encoder=encoder, unicode_map=unicode_map) + + +def _build_cp808() -> CodepageInfo: + base = resolve_codepage("cp866") + mapping = list(base.unicode_map) + mapping[0xFD - 0x80] = 0x20AC + encode_map = _build_encode_map(mapping) + + def encoder(text: str) -> bytes: + return _encode_with_map("cp808", text, encode_map) + + return CodepageInfo(canonical="cp808", encoder=encoder, unicode_map=tuple(mapping)) + + +def _build_kam() -> CodepageInfo: + base = resolve_codepage("cp437") + mapping = list(base.unicode_map) + overrides = { + 128: 0x010C, + 131: 0x010F, + 133: 0x010E, + 134: 0x0164, + 135: 0x010D, + 136: 0x011B, + 137: 0x011A, + 138: 0x0139, + 139: 0x00CD, + 140: 0x013E, + 141: 0x013A, + 143: 0x00C1, + 145: 0x017E, + 146: 0x017D, + 149: 0x00D3, + 150: 0x016F, + 151: 0x00DA, + 152: 0x00FD, + 155: 0x0160, + 156: 0x013D, + 157: 0x00DD, + 158: 0x0158, + 159: 0x0165, + 164: 0x0148, + 165: 0x0147, + 166: 0x016E, + 167: 0x00D4, + 168: 0x0161, + 169: 0x0159, + 170: 0x0155, + 171: 0x0154, + 173: 0x00A7, + } + for byte_value, codepoint in overrides.items(): + mapping[byte_value - 0x80] = codepoint + encode_map = _build_encode_map(mapping) + + def encoder(text: str) -> bytes: + return _encode_with_map("kam", text, encode_map) + + return CodepageInfo(canonical="kam", encoder=encoder, unicode_map=tuple(mapping)) + + +def _build_maz() -> CodepageInfo: + base = resolve_codepage("cp437") + mapping = list(base.unicode_map) + overrides = { + 134: 0x0105, + 141: 0x0107, + 143: 0x0104, + 144: 0x0118, + 145: 0x0119, + 146: 0x0142, + 149: 0x0106, + 152: 0x015A, + 156: 0x0141, + 158: 0x015B, + 160: 0x0179, + 161: 0x017B, + 163: 0x00D3, + 164: 0x0144, + 165: 0x0143, + 166: 0x017A, + 167: 0x017C, + } + for byte_value, codepoint in overrides.items(): + mapping[byte_value - 0x80] = codepoint + encode_map = _build_encode_map(mapping) + + def encoder(text: str) -> bytes: + return _encode_with_map("maz", text, encode_map) + + return CodepageInfo(canonical="maz", encoder=encoder, unicode_map=tuple(mapping)) + + +def _build_encode_map(mapping: List[int]) -> Dict[int, int]: + encode_map: Dict[int, int] = {} + for idx, codepoint in enumerate(mapping, start=128): + if codepoint >= 128 and codepoint not in encode_map: + encode_map[codepoint] = idx + return encode_map + + +def _encode_with_map(canonical: str, text: str, encode_map: Dict[int, int]) -> bytes: + result = bytearray() + for index, char in enumerate(text): + codepoint = ord(char) + if codepoint < 128: + result.append(codepoint) + continue + value = encode_map.get(codepoint) + if value is None: + raise UnicodeEncodeError(canonical, text, index, index + 1, "character not representable in codepage") + result.append(value) + return bytes(result) + + +def _encode_line(line: str, codepage: CodepageInfo) -> bytes: + return codepage.encode(f"{line}\n") + + +def _encoded_size(lines: List[str], codepage: CodepageInfo) -> int: + text = "\n".join(lines).rstrip("\n") + "\n" + return len(codepage.encode(text)) + + +def _unicode_map_bytes(codepage: CodepageInfo) -> bytes: + return b"".join(struct.pack(" bool: + return any(byte >= 0x80 for byte in data) @dataclass @@ -32,15 +260,24 @@ def main(argv: Iterable[str] | None = None) -> int: parser.add_argument("input", type=Path, help="Root Markdown file to convert.") parser.add_argument("output", type=Path, help="Output AMB filename.") parser.add_argument("--title", type=str, help="Optional book title.") + parser.add_argument( + "--codepage", + type=str, + default="437", + help="8-bit codepage for AMA text (default: 437 / cp437).", + ) args = parser.parse_args(list(argv) if argv is not None else None) input_path = args.input.resolve() if not input_path.exists(): parser.error(f"Input file '{input_path}' does not exist.") + codepage = resolve_codepage(args.codepage) + amb_bytes = build_amb( root_markdown=input_path, title=args.title, + codepage=codepage, ) args.output.parent.mkdir(parents=True, exist_ok=True) args.output.write_bytes(amb_bytes) @@ -48,10 +285,10 @@ def main(argv: Iterable[str] | None = None) -> int: return 0 -def build_amb(root_markdown: Path, title: str | None) -> bytes: +def build_amb(root_markdown: Path, title: str | None, codepage: CodepageInfo) -> bytes: articles = collect_articles(root_markdown) - ama_contents = render_articles(articles) - files = assemble_files(ama_contents, title) + ama_contents = render_articles(articles, codepage) + files = assemble_files(ama_contents, title, codepage) return pack_amb(files) @@ -114,7 +351,7 @@ def assign_ama_name(stem: str, existing: set[str]) -> str: return name -def render_articles(articles: Dict[Path, Article]) -> Dict[str, List[str]]: +def render_articles(articles: Dict[Path, Article], codepage: CodepageInfo) -> Dict[str, List[str]]: rendered: Dict[str, List[str]] = {} for path, article in articles.items(): @@ -128,7 +365,7 @@ def render_articles(articles: Dict[Path, Article]) -> Dict[str, List[str]]: base_path=path.parent, renderer_name="ama", ) - split_articles = split_article(article.ama_name, ama_lines) + split_articles = split_article(article.ama_name, ama_lines, codepage) rendered.update(split_articles) return rendered @@ -146,147 +383,129 @@ def rewrite_links(markdown: str, base_dir: Path, articles: Dict[Path, Article]) return MARKDOWN_LINK_RE.sub(replacer, markdown) -def split_article(filename: str, lines: List[str]) -> Dict[str, List[str]]: - def encoded_size(candidate: List[str]) -> int: - return len(("\n".join(candidate).rstrip("\n") + "\n").encode("utf-8")) - - if encoded_size(lines) <= AMA_MAX_BYTES: +def split_article(filename: str, lines: List[str], codepage: CodepageInfo) -> Dict[str, List[str]]: + if _encoded_size(lines, codepage) <= AMA_MAX_BYTES: return {filename: lines} - def line_size(value: str) -> int: - return len((value + "\n").encode("utf-8")) + encoded_lines: List[Tuple[str, bytes]] = [] + for idx, line in enumerate(lines): + try: + line_bytes = _encode_line(line, codepage) + except UnicodeEncodeError as exc: + raise ValueError( + f"Generated AMA article '{filename}' contains characters not representable in codepage '{codepage.canonical}' " + f"(line {idx + 1})." + ) from exc + if len(line_bytes) > AMA_MAX_BYTES: + raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.") + encoded_lines.append((line, line_bytes)) - segments: List[List[str]] = [] - segment_sizes: List[int] = [] - current: List[str] = [] + placeholder_target = "XXXXXXXX.XXX" + continue_overhead = len(_encode_line("", codepage)) + len( + _encode_line(f"%l{placeholder_target}:{LINK_CONTINUE_LABEL}%t", codepage) + ) + + segments: List[List[Tuple[str, bytes]]] = [] + current: List[Tuple[str, bytes]] = [] current_size = 0 + index = 0 - def flush_segment() -> None: - nonlocal current, current_size - if current: - segments.append(current) - segment_sizes.append(current_size) - current = [] - current_size = 0 - - for line in lines: - size = line_size(line) - if size > AMA_MAX_BYTES: + while index < len(encoded_lines): + line_text, line_bytes = encoded_lines[index] + line_length = len(line_bytes) + if current_size + line_length <= AMA_MAX_BYTES: + current.append((line_text, line_bytes)) + current_size += line_length + index += 1 + continue + if not current: raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.") - if current_size + size > AMA_MAX_BYTES: - flush_segment() - if current_size + size > AMA_MAX_BYTES: - raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.") - current.append(line) - current_size += size - flush_segment() + while current and current_size + continue_overhead > AMA_MAX_BYTES: + moved_line = current.pop() + current_size -= len(moved_line[1]) + encoded_lines.insert(index, moved_line) + + segments.append(current) + current = [] + current_size = 0 + + if current: + segments.append(current) + + segments = [segment for segment in segments if segment] if not segments: return {filename: lines} - soft_limit = AMA_MAX_BYTES - CONTINUE_OVERHEAD - idx = 0 - while idx < len(segments) - 1: - if not segments[idx]: - segments.pop(idx) - segment_sizes.pop(idx) - if idx > 0: - idx -= 1 - continue - if segment_sizes[idx] <= soft_limit: - idx += 1 - continue - moved_line = segments[idx].pop() - moved_size = line_size(moved_line) - segment_sizes[idx] -= moved_size - segments[idx + 1].insert(0, moved_line) - segment_sizes[idx + 1] += moved_size - - if not segments[idx]: - segments.pop(idx) - segment_sizes.pop(idx) - if idx > 0: - idx -= 1 - continue - - cascade = idx + 1 - while cascade < len(segments) and segment_sizes[cascade] > AMA_MAX_BYTES: - overflow_line = segments[cascade].pop() - overflow_size = line_size(overflow_line) - if overflow_size > AMA_MAX_BYTES: - raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.") - segment_sizes[cascade] -= overflow_size - if cascade + 1 < len(segments): - segments[cascade + 1].insert(0, overflow_line) - segment_sizes[cascade + 1] += overflow_size - else: - segments.append([overflow_line]) - segment_sizes.append(overflow_size) - if not segments[cascade]: - segments.pop(cascade) - segment_sizes.pop(cascade) - break - - # Recalculate segment sizes in case of structural changes - segment_sizes = [sum(line_size(line) for line in segment) for segment in segments] - - if len(segments) == 1: - return {filename: segments[0][:]} - stem = Path(filename).stem - result: Dict[str, List[str]] = {} generated_names: List[str] = [] existing_names: set[str] = set() - generated_names.append(filename) - existing_names.add(filename) - - for idx in range(1, len(segments)): - suffix = f"{idx:02d}" - trimmed = stem[: max(1, 8 - len(suffix))] - new_name = f"{trimmed}{suffix}.AMA" - counter = 1 - while new_name in existing_names: - suffix = f"{idx:02d}{counter}" + for idx in range(len(segments)): + if idx == 0: + new_name = filename + else: + suffix = f"{idx:02d}" trimmed = stem[: max(1, 8 - len(suffix))] new_name = f"{trimmed}{suffix}.AMA" - counter += 1 + counter = 1 + while new_name in existing_names: + suffix = f"{idx:02d}{counter}" + trimmed = stem[: max(1, 8 - len(suffix))] + new_name = f"{trimmed}{suffix}.AMA" + counter += 1 generated_names.append(new_name) existing_names.add(new_name) + result: Dict[str, List[str]] = {} for idx, name in enumerate(generated_names): - segment_lines = segments[idx][:] + segment_lines = [line for line, _ in segments[idx]] if idx < len(generated_names) - 1: segment_lines.append("") segment_lines.append(f"%l{generated_names[idx + 1]}:{LINK_CONTINUE_LABEL}%t") - if encoded_size(segment_lines) > AMA_MAX_BYTES: + if _encoded_size(segment_lines, codepage) > AMA_MAX_BYTES: raise ValueError(f"Unable to split AMA article '{name}' within size constraints.") result[name] = segment_lines return result -def assemble_files(ama_contents: Dict[str, List[str]], title: str | None) -> List[Tuple[str, bytes]]: +def assemble_files(ama_contents: Dict[str, List[str]], title: str | None, codepage: CodepageInfo) -> List[Tuple[str, bytes]]: files: List[Tuple[str, bytes]] = [] if title: files.append(("TITLE", title.encode("ascii", "ignore")[:64])) - index_bytes = encode_ama("INDEX.AMA", ama_contents.pop("INDEX.AMA")) + high_bit_used = False + + index_bytes = encode_ama("INDEX.AMA", ama_contents.pop("INDEX.AMA"), codepage) files.append(("INDEX.AMA", index_bytes)) + if _has_high_bit(index_bytes): + high_bit_used = True for name, lines in sorted(ama_contents.items()): - files.append((name, encode_ama(name, lines))) + data = encode_ama(name, lines, codepage) + files.append((name, data)) + if not high_bit_used and _has_high_bit(data): + high_bit_used = True + + if high_bit_used: + files.append(("UNICODE.MAP", _unicode_map_bytes(codepage))) return files -def encode_ama(name: str, lines: List[str]) -> bytes: - content = "\n".join(lines).rstrip("\n") + "\n" - data = content.encode("utf-8") - if len(data) > AMA_MAX_BYTES: - raise ValueError(f"Generated AMA article '{name}' exceeds {AMA_MAX_BYTES} bytes.") +def encode_ama(name: str, lines: List[str], codepage: CodepageInfo) -> bytes: if any("\t" in line for line in lines): raise ValueError(f"Generated AMA article '{name}' contains tab characters.") + content = "\n".join(lines).rstrip("\n") + "\n" + try: + data = codepage.encode(content) + except UnicodeEncodeError as exc: + raise ValueError( + f"Generated AMA article '{name}' contains characters not representable in codepage '{codepage.canonical}'." + ) from exc + if len(data) > AMA_MAX_BYTES: + raise ValueError(f"Generated AMA article '{name}' exceeds {AMA_MAX_BYTES} bytes.") return data