feat: parse markdown and custom directives into one representation

This commit is contained in:
randogoth 2026-10-04 19:21:48 +03:00
parent 3031b8acdd
commit e5583bbd89
8 changed files with 1410 additions and 2 deletions

655
core/src/parse.rs Normal file
View file

@ -0,0 +1,655 @@
//! Lowering Markdown into [`Doc`].
//!
//! `pulldown-cmark` is a pull parser whose events borrow from the source, which
//! maps directly onto the flat block sequence the formats want. The alternatives
//! build an arena AST that would only be flattened again.
//!
//! Only tables and strikethrough are enabled beyond CommonMark, matching what the
//! Python parsers actually handle. Events belonging to extensions that are off
//! are therefore never produced.
use std::path::Path;
use pulldown_cmark::{Alignment, CodeBlockKind, Event, Options, Parser, Tag, TagEnd};
use crate::error::Error;
use crate::ir::{Align, Block, Doc, Inline};
use crate::preprocess::{self, Segment};
/// Read a source file, expand its directives, and parse the result.
pub fn document(source: &Path, root: &Path) -> Result<Doc, Error> {
Ok(from_segments(&preprocess::expand(source, root)?))
}
/// Assemble a document from preprocessed segments.
///
/// A card break or an art block becomes a block directly, never passing through
/// the Markdown parser. That is what keeps a `{.card}` directive from surfacing
/// as literal paragraph text, which is a bug smolweb works around by stripping
/// the line before one of its two parsers sees it.
pub fn from_segments(segments: &[Segment]) -> Doc {
let mut doc = Doc { blocks: Vec::new(), first_h1: None };
for segment in segments {
match segment {
Segment::Markdown(text) => {
let mut parsed = markdown(text);
if doc.first_h1.is_none() {
doc.first_h1 = parsed.first_h1.take();
}
doc.blocks.append(&mut parsed.blocks);
}
Segment::Art { block_type, name, align, lines } => doc.blocks.push(Block::Art {
block_type: block_type.clone(),
name: name.clone(),
align: *align,
lines: lines.clone(),
}),
Segment::CardBreak { title } => {
doc.blocks.push(Block::CardBreak { title: title.clone() })
}
}
}
doc
}
/// Parse one run of Markdown.
pub fn markdown(text: &str) -> Doc {
let options = Options::ENABLE_TABLES | Options::ENABLE_STRIKETHROUGH;
let mut builder = Builder::default();
for event in Parser::new_ext(text, options) {
builder.handle(event);
}
builder.finish()
}
/// Data a tag needs when it closes, for the tags whose `TagEnd` carries none.
enum Pending {
Heading(u8),
Link { href: String, title: Option<String> },
Image { src: String, title: Option<String> },
}
struct ListFrame {
ordered: bool,
start: u64,
items: Vec<Vec<Block>>,
}
struct TableFrame {
alignments: Vec<Option<Align>>,
head: Vec<Vec<Inline>>,
rows: Vec<Vec<Vec<Inline>>>,
/// The row being filled, whether it belongs to the head or the body.
row: Vec<Vec<Inline>>,
in_head: bool,
}
struct CodeFrame {
info: Option<String>,
text: String,
}
struct Builder {
/// Stack of open block containers; index 0 is the document body.
blocks: Vec<Vec<Block>>,
/// Stack of open inline containers, for nested emphasis and link labels.
inlines: Vec<Vec<Inline>>,
pending: Vec<Pending>,
lists: Vec<ListFrame>,
tables: Vec<TableFrame>,
code: Option<CodeFrame>,
/// Accumulating the text of an open HTML block.
html: Option<String>,
/// Whether the innermost inline container was opened implicitly. A tight
/// list item holds its text with no `Paragraph` around it, so inline events
/// arrive with nothing open; without this they would be dropped.
implicit: bool,
first_h1: Option<String>,
}
impl Default for Builder {
fn default() -> Self {
Builder {
blocks: vec![Vec::new()],
inlines: Vec::new(),
pending: Vec::new(),
lists: Vec::new(),
tables: Vec::new(),
code: None,
html: None,
implicit: false,
first_h1: None,
}
}
}
impl Builder {
fn finish(mut self) -> Doc {
self.flush_implicit();
// Start and end events are balanced, so only the document frame is left.
let blocks = self.blocks.pop().unwrap_or_default();
Doc { blocks, first_h1: self.first_h1 }
}
fn handle(&mut self, event: Event<'_>) {
match event {
Event::Start(tag) => self.start(tag),
Event::End(end) => self.end(end),
Event::Text(text) => match &mut self.code {
Some(frame) => frame.text.push_str(&text),
None => self.push_inline(Inline::Text(text.to_string())),
},
Event::Code(code) => self.push_inline(Inline::Code(code.to_string())),
Event::SoftBreak => self.push_inline(Inline::SoftBreak),
Event::HardBreak => self.push_inline(Inline::HardBreak),
Event::Rule => self.push_block(Block::Rule),
Event::Html(html) => match &mut self.html {
Some(open) => open.push_str(&html),
None => self.push_block(Block::Html(html.to_string())),
},
Event::InlineHtml(html) => self.push_inline(Inline::Html(html.to_string())),
// Math, footnotes and task markers need extensions that are off.
_ => {}
}
}
fn start(&mut self, tag: Tag<'_>) {
// An implicit paragraph ends where the next block begins, so it has to
// be closed before that block is added or the order would invert.
if matches!(
tag,
Tag::Paragraph
| Tag::BlockQuote(_)
| Tag::CodeBlock(_)
| Tag::HtmlBlock
| Tag::List(_)
| Tag::Item
| Tag::Table(_)
) {
self.flush_implicit();
}
match tag {
Tag::Paragraph => self.open_inlines(),
Tag::Heading { level, .. } => {
self.pending.push(Pending::Heading(level as u8));
self.open_inlines();
}
Tag::BlockQuote(_) => self.blocks.push(Vec::new()),
Tag::CodeBlock(kind) => {
let info = match kind {
CodeBlockKind::Fenced(info) if !info.is_empty() => Some(info.to_string()),
_ => None,
};
self.code = Some(CodeFrame { info, text: String::new() });
}
Tag::HtmlBlock => self.html = Some(String::new()),
Tag::List(start) => self.lists.push(ListFrame {
ordered: start.is_some(),
start: start.unwrap_or(1),
items: Vec::new(),
}),
Tag::Item => self.blocks.push(Vec::new()),
Tag::Table(alignments) => self.tables.push(TableFrame {
alignments: alignments.into_iter().map(align_of).collect(),
head: Vec::new(),
rows: Vec::new(),
row: Vec::new(),
in_head: false,
}),
Tag::TableHead => {
if let Some(table) = self.tables.last_mut() {
table.in_head = true;
table.row.clear();
}
}
Tag::TableRow => {
if let Some(table) = self.tables.last_mut() {
table.row.clear();
}
}
Tag::TableCell => self.open_inlines(),
Tag::Emphasis | Tag::Strong | Tag::Strikethrough => self.open_inlines(),
Tag::Link { dest_url, title, .. } => {
self.pending
.push(Pending::Link { href: dest_url.to_string(), title: non_empty(&title) });
self.open_inlines();
}
Tag::Image { dest_url, title, .. } => {
self.pending
.push(Pending::Image { src: dest_url.to_string(), title: non_empty(&title) });
self.open_inlines();
}
_ => {}
}
}
fn end(&mut self, end: TagEnd) {
match end {
TagEnd::Paragraph => {
let inline = self.close_inlines();
self.push_block(Block::Paragraph(inline));
}
TagEnd::Heading(_) => {
let inline = self.close_inlines();
let level = match self.pending.pop() {
Some(Pending::Heading(level)) => level,
other => {
if let Some(value) = other {
self.pending.push(value);
}
1
}
};
if level == 1 && self.first_h1.is_none() {
self.first_h1 = Some(Doc::plain_text(&inline));
}
self.push_block(Block::Heading { level, inline });
}
TagEnd::BlockQuote(_) => {
self.flush_implicit();
let blocks = self.blocks.pop().unwrap_or_default();
self.push_block(Block::BlockQuote(blocks));
}
TagEnd::CodeBlock => {
if let Some(frame) = self.code.take() {
let lines = frame.text.lines().map(str::to_string).collect();
self.push_block(Block::CodeBlock { info: frame.info, lines });
}
}
TagEnd::HtmlBlock => {
if let Some(html) = self.html.take() {
self.push_block(Block::Html(html));
}
}
TagEnd::List(_) => {
self.flush_implicit();
if let Some(frame) = self.lists.pop() {
self.push_block(Block::List {
ordered: frame.ordered,
start: frame.start,
items: frame.items,
});
}
}
TagEnd::Item => {
self.flush_implicit();
let item = self.blocks.pop().unwrap_or_default();
if let Some(list) = self.lists.last_mut() {
list.items.push(item);
}
}
TagEnd::Table => {
if let Some(frame) = self.tables.pop() {
self.push_block(Block::Table {
alignments: frame.alignments,
head: frame.head,
rows: frame.rows,
});
}
}
TagEnd::TableHead => {
if let Some(table) = self.tables.last_mut() {
table.head = std::mem::take(&mut table.row);
table.in_head = false;
}
}
TagEnd::TableRow => {
if let Some(table) = self.tables.last_mut() {
let row = std::mem::take(&mut table.row);
table.rows.push(row);
}
}
TagEnd::TableCell => {
let cell = self.close_inlines();
if let Some(table) = self.tables.last_mut() {
table.row.push(cell);
}
}
TagEnd::Emphasis => {
let inner = self.close_inlines();
self.push_inline(Inline::Emph(inner));
}
TagEnd::Strong => {
let inner = self.close_inlines();
self.push_inline(Inline::Strong(inner));
}
TagEnd::Strikethrough => {
let inner = self.close_inlines();
self.push_inline(Inline::Strike(inner));
}
TagEnd::Link => {
let label = self.close_inlines();
if let Some(Pending::Link { href, title }) = self.pending.pop() {
self.push_inline(Inline::Link { href, title, label });
}
}
TagEnd::Image => {
let alt = self.close_inlines();
if let Some(Pending::Image { src, title }) = self.pending.pop() {
self.push_inline(Inline::Image { src, title, alt });
}
}
_ => {}
}
}
fn open_inlines(&mut self) {
self.inlines.push(Vec::new());
}
fn close_inlines(&mut self) -> Vec<Inline> {
self.inlines.pop().unwrap_or_default()
}
/// Add an inline to the innermost open inline container, opening an implicit
/// paragraph if nothing is open. That happens for every tight list item,
/// whose text pulldown-cmark emits with no `Paragraph` around it.
fn push_inline(&mut self, inline: Inline) {
if self.inlines.is_empty() {
self.inlines.push(Vec::new());
self.implicit = true;
}
if let Some(frame) = self.inlines.last_mut() {
frame.push(inline);
}
}
/// Close an implicit paragraph, if one is open, into the current block
/// container. Pushes directly rather than through `push_block`, which calls
/// this.
fn flush_implicit(&mut self) {
if !self.implicit {
return;
}
self.implicit = false;
let inline = self.inlines.pop().unwrap_or_default();
if !inline.is_empty()
&& let Some(frame) = self.blocks.last_mut()
{
frame.push(Block::Paragraph(inline));
}
}
fn push_block(&mut self, block: Block) {
self.flush_implicit();
if let Some(frame) = self.blocks.last_mut() {
frame.push(block);
}
}
}
fn align_of(alignment: Alignment) -> Option<Align> {
match alignment {
Alignment::None => None,
Alignment::Left => Some(Align::Left),
Alignment::Center => Some(Align::Center),
Alignment::Right => Some(Align::Right),
}
}
fn non_empty(value: &str) -> Option<String> {
(!value.is_empty()).then(|| value.to_string())
}
#[cfg(test)]
mod tests {
use super::*;
fn blocks(text: &str) -> Vec<Block> {
markdown(text).blocks
}
fn text(value: &str) -> Vec<Inline> {
vec![Inline::Text(value.into())]
}
#[test]
fn parses_headings_and_paragraphs() {
assert_eq!(
blocks("# Title\n\nBody.\n"),
vec![
Block::Heading { level: 1, inline: text("Title") },
Block::Paragraph(text("Body.")),
]
);
}
#[test]
fn parses_setext_headings() {
// wapdown's parser handles these and md2txt's does not, so unifying the
// two parsers gains them for the text formats.
assert_eq!(
blocks("Title\n=====\n"),
vec![Block::Heading { level: 1, inline: text("Title") }]
);
}
#[test]
fn records_the_first_level_one_heading() {
let doc = markdown("## Second\n\n# First\n\n# Another\n");
assert_eq!(doc.first_h1.as_deref(), Some("First"));
// A document with no h1 has nothing to derive a title from.
assert_eq!(markdown("## Only\n").first_h1, None);
}
#[test]
fn parses_inline_markup() {
let Block::Paragraph(inline) = &blocks("a *b* **c** ~~d~~ `e`\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(
inline,
&vec![
Inline::Text("a ".into()),
Inline::Emph(text("b")),
Inline::Text(" ".into()),
Inline::Strong(text("c")),
Inline::Text(" ".into()),
Inline::Strike(text("d")),
Inline::Text(" ".into()),
Inline::Code("e".into()),
]
);
}
#[test]
fn parses_links_and_images_with_their_labels() {
let Block::Paragraph(inline) = &blocks("[a *b*](/x \"T\") ![alt](/i.png)\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(
inline[0],
Inline::Link {
href: "/x".into(),
title: Some("T".into()),
label: vec![Inline::Text("a ".into()), Inline::Emph(text("b"))],
}
);
assert_eq!(
inline[2],
Inline::Image { src: "/i.png".into(), title: None, alt: text("alt") }
);
}
#[test]
fn parses_nested_lists() {
let parsed = blocks("- a\n - b\n- c\n");
let Block::List { ordered, start, items } = &parsed[0] else { panic!("expected a list") };
assert!(!ordered);
assert_eq!(*start, 1);
assert_eq!(items.len(), 2);
// The first item holds its own paragraph and the nested list.
assert_eq!(items[0][0], Block::Paragraph(text("a")));
assert!(matches!(items[0][1], Block::List { .. }));
assert_eq!(items[1][0], Block::Paragraph(text("c")));
}
#[test]
fn a_tight_list_items_text_is_kept() {
// Regression: pulldown-cmark emits a tight item's text with no Paragraph
// around it, so an implicit one has to be opened or the text is dropped.
let Block::List { items, .. } = &blocks("- a\n- b\n")[0] else { panic!("expected a list") };
assert_eq!(
items,
&vec![vec![Block::Paragraph(text("a"))], vec![Block::Paragraph(text("b"))]]
);
}
#[test]
fn a_tight_item_keeps_its_inline_markup_in_order() {
let Block::List { items, .. } = &blocks("- a *b* c\n")[0] else {
panic!("expected a list")
};
assert_eq!(
items[0],
vec![Block::Paragraph(vec![
Inline::Text("a ".into()),
Inline::Emph(text("b")),
Inline::Text(" c".into()),
])]
);
}
#[test]
fn a_tight_item_opening_with_markup_is_still_one_paragraph() {
let Block::List { items, .. } = &blocks("- *a* b\n")[0] else { panic!("expected a list") };
assert_eq!(
items[0],
vec![Block::Paragraph(vec![Inline::Emph(text("a")), Inline::Text(" b".into()),])]
);
}
#[test]
fn a_loose_list_is_shaped_the_same_as_a_tight_one() {
// The distinction is pulldown-cmark's, not ours: both yield a paragraph
// per item, so no format has to know which it was.
let Block::List { items: tight, .. } = &blocks("- a\n- b\n")[0] else { panic!("list") };
let Block::List { items: loose, .. } = &blocks("- a\n\n- b\n")[0] else { panic!("list") };
assert_eq!(tight, loose);
}
#[test]
fn an_ordered_list_keeps_its_starting_number() {
let Block::List { ordered, start, items } = &blocks("3. a\n4. b\n")[0] else {
panic!("expected a list")
};
assert!(ordered);
assert_eq!(*start, 3);
assert_eq!(items.len(), 2);
}
#[test]
fn parses_nested_blockquotes() {
// Depth comes from the nesting, so there is one place it is recorded.
let parsed = blocks("> outer\n>\n> > inner\n");
let Block::BlockQuote(outer) = &parsed[0] else { panic!("expected a quote") };
assert_eq!(outer[0], Block::Paragraph(text("outer")));
assert_eq!(outer[1], Block::BlockQuote(vec![Block::Paragraph(text("inner"))]));
}
#[test]
fn parses_fenced_and_indented_code() {
assert_eq!(
blocks("```rust\nlet x = 1;\nlet y = 2;\n```\n"),
vec![Block::CodeBlock {
info: Some("rust".into()),
lines: vec!["let x = 1;".into(), "let y = 2;".into()],
}]
);
assert_eq!(
blocks(" indented\n"),
vec![Block::CodeBlock { info: None, lines: vec!["indented".into()] }]
);
}
#[test]
fn parses_tables_with_their_alignments() {
let parsed = blocks("| a | b |\n| :-- | --: |\n| 1 | 2 |\n| 3 | 4 |\n");
let Block::Table { alignments, head, rows } = &parsed[0] else {
panic!("expected a table")
};
assert_eq!(alignments, &vec![Some(Align::Left), Some(Align::Right)]);
assert_eq!(head, &vec![text("a"), text("b")]);
assert_eq!(rows, &vec![vec![text("1"), text("2")], vec![text("3"), text("4")]]);
}
#[test]
fn parses_rules_and_breaks() {
assert_eq!(blocks("---\n"), vec![Block::Rule]);
let Block::Paragraph(inline) = &blocks("a\nb\n")[0] else { panic!("expected a paragraph") };
assert_eq!(inline[1], Inline::SoftBreak);
let Block::Paragraph(inline) = &blocks("a \nb\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(inline[1], Inline::HardBreak);
}
#[test]
fn keeps_raw_html_rather_than_dropping_it() {
// XHTML-MP passes it through and the text formats strip it; the parser
// does not get to decide.
let parsed = blocks("<div>x</div>\n");
assert!(matches!(&parsed[0], Block::Html(html) if html.contains("<div>")));
let Block::Paragraph(inline) = &blocks("a <b>c</b>\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(inline[1], Inline::Html("<b>".into()));
}
// -- Segment assembly --------------------------------------------------
#[test]
fn card_breaks_and_art_become_blocks_without_passing_through_the_parser() {
let segments = vec![
Segment::Markdown("Intro.\n".into()),
Segment::CardBreak { title: Some("Weather".into()) },
Segment::Markdown("Cold.\n".into()),
Segment::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec!["/\\".into()],
},
];
let doc = from_segments(&segments);
assert_eq!(
doc.blocks,
vec![
Block::Paragraph(text("Intro.")),
Block::CardBreak { title: Some("Weather".into()) },
Block::Paragraph(text("Cold.")),
Block::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec!["/\\".into()],
},
]
);
}
#[test]
fn a_card_directive_never_appears_as_text() {
// The leak smolweb works around by stripping the line before md2txt sees
// it: here the directive cannot reach the Markdown parser at all.
let dir = tempfile::tempdir().unwrap();
let root = dir.path().canonicalize().unwrap();
std::fs::write(root.join("page.md"), "Intro.\n\n{.card Weather}\nCold.\n").unwrap();
let doc = document(&root.join("page.md"), &root).unwrap();
let rendered = format!("{:?}", doc.blocks);
assert!(!rendered.contains("{.card"), "{rendered}");
assert!(doc.blocks.contains(&Block::CardBreak { title: Some("Weather".into()) }));
}
#[test]
fn the_title_comes_from_the_first_markdown_run_that_has_one() {
let segments = vec![
Segment::CardBreak { title: None },
Segment::Markdown("## not it\n".into()),
Segment::Markdown("# Found\n".into()),
Segment::Markdown("# Later\n".into()),
];
assert_eq!(from_segments(&segments).first_h1.as_deref(), Some("Found"));
}
}