This commit is contained in:
randogoth 2025-10-20 21:43:08 +03:00
parent 6350a8a305
commit 2db4eab94c
2 changed files with 328 additions and 107 deletions

View file

@ -2,12 +2,13 @@
from __future__ import annotations
import argparse
import codecs
import re
import struct
from collections import deque
from dataclasses import dataclass
from pathlib import Path
from typing import Dict, Iterable, List, Tuple
from typing import Callable, Dict, Iterable, List, Tuple
from md2txt import convert_markdown
from md2txt.conversion.core import parse_frontmatter
@ -18,7 +19,234 @@ EXT_MD = {".md", ".markdown", ".mkd", ".mkdn"}
AMA_MAX_BYTES = 65_535
AMB_MAGIC = b"AMB1"
LINK_CONTINUE_LABEL = "Continue"
CONTINUE_OVERHEAD = len("\n".encode("utf-8")) + len((f"%l{'ABCDEFGH.AMA'}:{LINK_CONTINUE_LABEL}%t\n").encode("utf-8"))
@dataclass(frozen=True)
class CodepageInfo:
canonical: str
encoder: Callable[[str], bytes]
unicode_map: Tuple[int, ...]
def encode(self, text: str) -> bytes:
return self.encoder(text)
CODEPAGE_ALIASES: Dict[str, str] = {
"cp437": "cp437",
"ibm437": "cp437",
"dos437": "cp437",
"437": "cp437",
"cp775": "cp775",
"775": "cp775",
"cp808": "cp808",
"808": "cp808",
"cp850": "cp850",
"850": "cp850",
"cp852": "cp852",
"852": "cp852",
"cp857": "cp857",
"857": "cp857",
"cp858": "cp858",
"858": "cp858",
"cp866": "cp866",
"866": "cp866",
"cp1250": "cp1250",
"1250": "cp1250",
"windows1250": "cp1250",
"win1250": "cp1250",
"cp1252": "cp1252",
"1252": "cp1252",
"windows1252": "cp1252",
"win1252": "cp1252",
"kam": "kam",
"kamenicky": "kam",
"kamenickyencoding": "kam",
"maz": "maz",
"mazovia": "maz",
}
CODEPAGE_CACHE: Dict[str, CodepageInfo] = {}
def resolve_codepage(name: str) -> CodepageInfo:
normalized = _normalize_codepage_name(name)
try:
return CODEPAGE_CACHE[normalized]
except KeyError:
info = _build_codepage(normalized)
CODEPAGE_CACHE[normalized] = info
return info
def _normalize_codepage_name(raw: str) -> str:
token = raw.strip().lower()
token = token.replace("-", "").replace("_", "")
if token in CODEPAGE_ALIASES:
return CODEPAGE_ALIASES[token]
if token.startswith("ibm") and token[3:].isdigit():
return f"cp{token[3:]}"
if token.startswith("dos") and token[3:].isdigit():
return f"cp{token[3:]}"
if token.startswith("windows") and token[7:].isdigit():
return f"cp{token[7:]}"
if token.startswith("win") and token[3:].isdigit():
return f"cp{token[3:]}"
if token.isdigit():
return f"cp{token}"
return token
def _build_codepage(canonical: str) -> CodepageInfo:
if canonical == "cp808":
return _build_cp808()
if canonical == "kam":
return _build_kam()
if canonical == "maz":
return _build_maz()
try:
codec_info = codecs.lookup(canonical)
except LookupError as exc:
raise ValueError(f"Unsupported codepage '{canonical}'.") from exc
def encoder(text: str) -> bytes:
return text.encode(codec_info.name, "strict")
try:
high_bytes = bytes(range(128, 256)).decode(codec_info.name, "strict")
except UnicodeDecodeError as exc:
raise ValueError(f"Codepage '{canonical}' is not an 8-bit single-byte encoding.") from exc
unicode_map = tuple(ord(ch) for ch in high_bytes)
return CodepageInfo(canonical=codec_info.name, encoder=encoder, unicode_map=unicode_map)
def _build_cp808() -> CodepageInfo:
base = resolve_codepage("cp866")
mapping = list(base.unicode_map)
mapping[0xFD - 0x80] = 0x20AC
encode_map = _build_encode_map(mapping)
def encoder(text: str) -> bytes:
return _encode_with_map("cp808", text, encode_map)
return CodepageInfo(canonical="cp808", encoder=encoder, unicode_map=tuple(mapping))
def _build_kam() -> CodepageInfo:
base = resolve_codepage("cp437")
mapping = list(base.unicode_map)
overrides = {
128: 0x010C,
131: 0x010F,
133: 0x010E,
134: 0x0164,
135: 0x010D,
136: 0x011B,
137: 0x011A,
138: 0x0139,
139: 0x00CD,
140: 0x013E,
141: 0x013A,
143: 0x00C1,
145: 0x017E,
146: 0x017D,
149: 0x00D3,
150: 0x016F,
151: 0x00DA,
152: 0x00FD,
155: 0x0160,
156: 0x013D,
157: 0x00DD,
158: 0x0158,
159: 0x0165,
164: 0x0148,
165: 0x0147,
166: 0x016E,
167: 0x00D4,
168: 0x0161,
169: 0x0159,
170: 0x0155,
171: 0x0154,
173: 0x00A7,
}
for byte_value, codepoint in overrides.items():
mapping[byte_value - 0x80] = codepoint
encode_map = _build_encode_map(mapping)
def encoder(text: str) -> bytes:
return _encode_with_map("kam", text, encode_map)
return CodepageInfo(canonical="kam", encoder=encoder, unicode_map=tuple(mapping))
def _build_maz() -> CodepageInfo:
base = resolve_codepage("cp437")
mapping = list(base.unicode_map)
overrides = {
134: 0x0105,
141: 0x0107,
143: 0x0104,
144: 0x0118,
145: 0x0119,
146: 0x0142,
149: 0x0106,
152: 0x015A,
156: 0x0141,
158: 0x015B,
160: 0x0179,
161: 0x017B,
163: 0x00D3,
164: 0x0144,
165: 0x0143,
166: 0x017A,
167: 0x017C,
}
for byte_value, codepoint in overrides.items():
mapping[byte_value - 0x80] = codepoint
encode_map = _build_encode_map(mapping)
def encoder(text: str) -> bytes:
return _encode_with_map("maz", text, encode_map)
return CodepageInfo(canonical="maz", encoder=encoder, unicode_map=tuple(mapping))
def _build_encode_map(mapping: List[int]) -> Dict[int, int]:
encode_map: Dict[int, int] = {}
for idx, codepoint in enumerate(mapping, start=128):
if codepoint >= 128 and codepoint not in encode_map:
encode_map[codepoint] = idx
return encode_map
def _encode_with_map(canonical: str, text: str, encode_map: Dict[int, int]) -> bytes:
result = bytearray()
for index, char in enumerate(text):
codepoint = ord(char)
if codepoint < 128:
result.append(codepoint)
continue
value = encode_map.get(codepoint)
if value is None:
raise UnicodeEncodeError(canonical, text, index, index + 1, "character not representable in codepage")
result.append(value)
return bytes(result)
def _encode_line(line: str, codepage: CodepageInfo) -> bytes:
return codepage.encode(f"{line}\n")
def _encoded_size(lines: List[str], codepage: CodepageInfo) -> int:
text = "\n".join(lines).rstrip("\n") + "\n"
return len(codepage.encode(text))
def _unicode_map_bytes(codepage: CodepageInfo) -> bytes:
return b"".join(struct.pack("<H", value) for value in codepage.unicode_map)
def _has_high_bit(data: bytes) -> bool:
return any(byte >= 0x80 for byte in data)
@dataclass
@ -32,15 +260,24 @@ def main(argv: Iterable[str] | None = None) -> int:
parser.add_argument("input", type=Path, help="Root Markdown file to convert.")
parser.add_argument("output", type=Path, help="Output AMB filename.")
parser.add_argument("--title", type=str, help="Optional book title.")
parser.add_argument(
"--codepage",
type=str,
default="437",
help="8-bit codepage for AMA text (default: 437 / cp437).",
)
args = parser.parse_args(list(argv) if argv is not None else None)
input_path = args.input.resolve()
if not input_path.exists():
parser.error(f"Input file '{input_path}' does not exist.")
codepage = resolve_codepage(args.codepage)
amb_bytes = build_amb(
root_markdown=input_path,
title=args.title,
codepage=codepage,
)
args.output.parent.mkdir(parents=True, exist_ok=True)
args.output.write_bytes(amb_bytes)
@ -48,10 +285,10 @@ def main(argv: Iterable[str] | None = None) -> int:
return 0
def build_amb(root_markdown: Path, title: str | None) -> bytes:
def build_amb(root_markdown: Path, title: str | None, codepage: CodepageInfo) -> bytes:
articles = collect_articles(root_markdown)
ama_contents = render_articles(articles)
files = assemble_files(ama_contents, title)
ama_contents = render_articles(articles, codepage)
files = assemble_files(ama_contents, title, codepage)
return pack_amb(files)
@ -114,7 +351,7 @@ def assign_ama_name(stem: str, existing: set[str]) -> str:
return name
def render_articles(articles: Dict[Path, Article]) -> Dict[str, List[str]]:
def render_articles(articles: Dict[Path, Article], codepage: CodepageInfo) -> Dict[str, List[str]]:
rendered: Dict[str, List[str]] = {}
for path, article in articles.items():
@ -128,7 +365,7 @@ def render_articles(articles: Dict[Path, Article]) -> Dict[str, List[str]]:
base_path=path.parent,
renderer_name="ama",
)
split_articles = split_article(article.ama_name, ama_lines)
split_articles = split_article(article.ama_name, ama_lines, codepage)
rendered.update(split_articles)
return rendered
@ -146,147 +383,129 @@ def rewrite_links(markdown: str, base_dir: Path, articles: Dict[Path, Article])
return MARKDOWN_LINK_RE.sub(replacer, markdown)
def split_article(filename: str, lines: List[str]) -> Dict[str, List[str]]:
def encoded_size(candidate: List[str]) -> int:
return len(("\n".join(candidate).rstrip("\n") + "\n").encode("utf-8"))
if encoded_size(lines) <= AMA_MAX_BYTES:
def split_article(filename: str, lines: List[str], codepage: CodepageInfo) -> Dict[str, List[str]]:
if _encoded_size(lines, codepage) <= AMA_MAX_BYTES:
return {filename: lines}
def line_size(value: str) -> int:
return len((value + "\n").encode("utf-8"))
encoded_lines: List[Tuple[str, bytes]] = []
for idx, line in enumerate(lines):
try:
line_bytes = _encode_line(line, codepage)
except UnicodeEncodeError as exc:
raise ValueError(
f"Generated AMA article '{filename}' contains characters not representable in codepage '{codepage.canonical}' "
f"(line {idx + 1})."
) from exc
if len(line_bytes) > AMA_MAX_BYTES:
raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.")
encoded_lines.append((line, line_bytes))
segments: List[List[str]] = []
segment_sizes: List[int] = []
current: List[str] = []
placeholder_target = "XXXXXXXX.XXX"
continue_overhead = len(_encode_line("", codepage)) + len(
_encode_line(f"%l{placeholder_target}:{LINK_CONTINUE_LABEL}%t", codepage)
)
segments: List[List[Tuple[str, bytes]]] = []
current: List[Tuple[str, bytes]] = []
current_size = 0
index = 0
def flush_segment() -> None:
nonlocal current, current_size
if current:
segments.append(current)
segment_sizes.append(current_size)
current = []
current_size = 0
for line in lines:
size = line_size(line)
if size > AMA_MAX_BYTES:
while index < len(encoded_lines):
line_text, line_bytes = encoded_lines[index]
line_length = len(line_bytes)
if current_size + line_length <= AMA_MAX_BYTES:
current.append((line_text, line_bytes))
current_size += line_length
index += 1
continue
if not current:
raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.")
if current_size + size > AMA_MAX_BYTES:
flush_segment()
if current_size + size > AMA_MAX_BYTES:
raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.")
current.append(line)
current_size += size
flush_segment()
while current and current_size + continue_overhead > AMA_MAX_BYTES:
moved_line = current.pop()
current_size -= len(moved_line[1])
encoded_lines.insert(index, moved_line)
segments.append(current)
current = []
current_size = 0
if current:
segments.append(current)
segments = [segment for segment in segments if segment]
if not segments:
return {filename: lines}
soft_limit = AMA_MAX_BYTES - CONTINUE_OVERHEAD
idx = 0
while idx < len(segments) - 1:
if not segments[idx]:
segments.pop(idx)
segment_sizes.pop(idx)
if idx > 0:
idx -= 1
continue
if segment_sizes[idx] <= soft_limit:
idx += 1
continue
moved_line = segments[idx].pop()
moved_size = line_size(moved_line)
segment_sizes[idx] -= moved_size
segments[idx + 1].insert(0, moved_line)
segment_sizes[idx + 1] += moved_size
if not segments[idx]:
segments.pop(idx)
segment_sizes.pop(idx)
if idx > 0:
idx -= 1
continue
cascade = idx + 1
while cascade < len(segments) and segment_sizes[cascade] > AMA_MAX_BYTES:
overflow_line = segments[cascade].pop()
overflow_size = line_size(overflow_line)
if overflow_size > AMA_MAX_BYTES:
raise ValueError(f"Generated AMA article '{filename}' contains a line exceeding {AMA_MAX_BYTES} bytes.")
segment_sizes[cascade] -= overflow_size
if cascade + 1 < len(segments):
segments[cascade + 1].insert(0, overflow_line)
segment_sizes[cascade + 1] += overflow_size
else:
segments.append([overflow_line])
segment_sizes.append(overflow_size)
if not segments[cascade]:
segments.pop(cascade)
segment_sizes.pop(cascade)
break
# Recalculate segment sizes in case of structural changes
segment_sizes = [sum(line_size(line) for line in segment) for segment in segments]
if len(segments) == 1:
return {filename: segments[0][:]}
stem = Path(filename).stem
result: Dict[str, List[str]] = {}
generated_names: List[str] = []
existing_names: set[str] = set()
generated_names.append(filename)
existing_names.add(filename)
for idx in range(1, len(segments)):
suffix = f"{idx:02d}"
trimmed = stem[: max(1, 8 - len(suffix))]
new_name = f"{trimmed}{suffix}.AMA"
counter = 1
while new_name in existing_names:
suffix = f"{idx:02d}{counter}"
for idx in range(len(segments)):
if idx == 0:
new_name = filename
else:
suffix = f"{idx:02d}"
trimmed = stem[: max(1, 8 - len(suffix))]
new_name = f"{trimmed}{suffix}.AMA"
counter += 1
counter = 1
while new_name in existing_names:
suffix = f"{idx:02d}{counter}"
trimmed = stem[: max(1, 8 - len(suffix))]
new_name = f"{trimmed}{suffix}.AMA"
counter += 1
generated_names.append(new_name)
existing_names.add(new_name)
result: Dict[str, List[str]] = {}
for idx, name in enumerate(generated_names):
segment_lines = segments[idx][:]
segment_lines = [line for line, _ in segments[idx]]
if idx < len(generated_names) - 1:
segment_lines.append("")
segment_lines.append(f"%l{generated_names[idx + 1]}:{LINK_CONTINUE_LABEL}%t")
if encoded_size(segment_lines) > AMA_MAX_BYTES:
if _encoded_size(segment_lines, codepage) > AMA_MAX_BYTES:
raise ValueError(f"Unable to split AMA article '{name}' within size constraints.")
result[name] = segment_lines
return result
def assemble_files(ama_contents: Dict[str, List[str]], title: str | None) -> List[Tuple[str, bytes]]:
def assemble_files(ama_contents: Dict[str, List[str]], title: str | None, codepage: CodepageInfo) -> List[Tuple[str, bytes]]:
files: List[Tuple[str, bytes]] = []
if title:
files.append(("TITLE", title.encode("ascii", "ignore")[:64]))
index_bytes = encode_ama("INDEX.AMA", ama_contents.pop("INDEX.AMA"))
high_bit_used = False
index_bytes = encode_ama("INDEX.AMA", ama_contents.pop("INDEX.AMA"), codepage)
files.append(("INDEX.AMA", index_bytes))
if _has_high_bit(index_bytes):
high_bit_used = True
for name, lines in sorted(ama_contents.items()):
files.append((name, encode_ama(name, lines)))
data = encode_ama(name, lines, codepage)
files.append((name, data))
if not high_bit_used and _has_high_bit(data):
high_bit_used = True
if high_bit_used:
files.append(("UNICODE.MAP", _unicode_map_bytes(codepage)))
return files
def encode_ama(name: str, lines: List[str]) -> bytes:
content = "\n".join(lines).rstrip("\n") + "\n"
data = content.encode("utf-8")
if len(data) > AMA_MAX_BYTES:
raise ValueError(f"Generated AMA article '{name}' exceeds {AMA_MAX_BYTES} bytes.")
def encode_ama(name: str, lines: List[str], codepage: CodepageInfo) -> bytes:
if any("\t" in line for line in lines):
raise ValueError(f"Generated AMA article '{name}' contains tab characters.")
content = "\n".join(lines).rstrip("\n") + "\n"
try:
data = codepage.encode(content)
except UnicodeEncodeError as exc:
raise ValueError(
f"Generated AMA article '{name}' contains characters not representable in codepage '{codepage.canonical}'."
) from exc
if len(data) > AMA_MAX_BYTES:
raise ValueError(f"Generated AMA article '{name}' exceeds {AMA_MAX_BYTES} bytes.")
return data