feat: parse markdown and custom directives into one representation

This commit is contained in:
randogoth 2026-10-04 19:21:48 +03:00
parent 3031b8acdd
commit e5583bbd89
8 changed files with 1410 additions and 2 deletions

View file

@ -8,6 +8,7 @@ license = "Apache-2.0"
publish = false
[dependencies]
pulldown-cmark = { version = "0.13", default-features = false }
serde = { version = "1.0", features = ["derive"] }
toml = "1.1"

View file

@ -25,6 +25,37 @@ pub enum Error {
/// the registry's full id set, which is what tells an operator whether the
/// name is a typo or a missing cargo feature.
UnknownFormat { listener: String, format: String, available: Vec<String> },
/// An include or ASCII-art target cannot be used. `path` is where the
/// directive actually pointed, resolved, which is the thing an author needs
/// to see when a relative target is wrong.
Include { path: PathBuf, reason: IncludeReason },
}
/// Why a directive's target was refused.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum IncludeReason {
/// Resolved outside the content root, or is not a regular file.
Outside,
Missing,
/// Already being expanded further up the stack.
Circular,
TooDeep,
/// The expansion exceeded its line or byte budget.
TooLarge,
}
impl fmt::Display for IncludeReason {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
let text = match self {
IncludeReason::Outside => "is outside the content root",
IncludeReason::Missing => "does not exist",
IncludeReason::Circular => "is already being included",
IncludeReason::TooDeep => "includes nest too deeply",
IncludeReason::TooLarge => "expands to more than the size limit",
};
f.write_str(text)
}
}
impl fmt::Display for Error {
@ -37,6 +68,9 @@ impl fmt::Display for Error {
Error::DuplicateHost { host, first, second } => {
write!(f, "host '{host}' is claimed by both site '{first}' and site '{second}'")
}
Error::Include { path, reason } => {
write!(f, "include target {} {reason}", path.display())
}
Error::UnknownFormat { listener, format, available } => write!(
f,
"listener '{listener}' wants format '{format}', which this build does not \

142
core/src/ir.rs Normal file
View file

@ -0,0 +1,142 @@
//! The one parsed representation every output format consumes.
//!
//! smolweb parses each document twice, with two hand-rolled regex parsers that
//! share seven identical patterns but disagree on the edges: that is where the
//! `{.card}` directive leaks into gemtext as literal text, and why the two
//! libraries carry two divergent sets of defaults. Parsing once into this
//! structure removes the class of bug rather than the instances.
//!
//! It is a flat block sequence rather than a tree of nodes because that is what
//! the consumers want: WML packs a linear run of blocks into byte-budgeted
//! cards, and the text renderers are a fold over blocks.
/// A parsed document.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Doc {
pub blocks: Vec<Block>,
/// The first level-1 heading as plain text, used to derive a page title when
/// the directory config does not set one.
pub first_h1: Option<String>,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Block {
Paragraph(Vec<Inline>),
Heading {
level: u8,
inline: Vec<Inline>,
},
/// `info` is the fence's info string, so a renderer can label or highlight.
CodeBlock {
info: Option<String>,
lines: Vec<String>,
},
/// Nesting carries the depth, so there is one place it is recorded.
BlockQuote(Vec<Block>),
List {
ordered: bool,
start: u64,
items: Vec<Vec<Block>>,
},
Rule,
Table {
alignments: Vec<Option<Align>>,
head: Vec<Vec<Inline>>,
rows: Vec<Vec<Vec<Inline>>>,
},
/// A verbatim block lifted from a file by `#[label](art.txt)`. Already read,
/// because nothing downstream of the parser touches the filesystem.
Art {
block_type: String,
name: String,
align: Option<Align>,
lines: Vec<String>,
},
/// A `{.card Title}` divider. Formats that paginate start a new screen here;
/// every other format emits nothing at all for it.
CardBreak {
title: Option<String>,
},
/// Raw block HTML. Kept rather than dropped so XHTML-MP can pass it through
/// and the text formats can strip it, instead of the parser deciding.
Html(String),
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Inline {
Text(String),
Code(String),
Emph(Vec<Inline>),
Strong(Vec<Inline>),
Strike(Vec<Inline>),
Link {
href: String,
title: Option<String>,
label: Vec<Inline>,
},
Image {
src: String,
title: Option<String>,
alt: Vec<Inline>,
},
/// A line break the source left soft; renderers that wrap ignore it.
SoftBreak,
HardBreak,
Html(String),
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Align {
Left,
Center,
Right,
}
impl Doc {
/// Plain text of a run of inlines, with markup dropped and link labels kept.
/// Used for the title fallback and anywhere a format needs a bare string.
pub fn plain_text(inline: &[Inline]) -> String {
let mut out = String::new();
Self::write_plain(inline, &mut out);
out
}
fn write_plain(inline: &[Inline], out: &mut String) {
for item in inline {
match item {
Inline::Text(text) | Inline::Code(text) => out.push_str(text),
Inline::Emph(inner) | Inline::Strong(inner) | Inline::Strike(inner) => {
Self::write_plain(inner, out)
}
Inline::Link { label, .. } => Self::write_plain(label, out),
Inline::Image { alt, .. } => Self::write_plain(alt, out),
Inline::SoftBreak | Inline::HardBreak => out.push(' '),
// Raw markup is not text; a format that wants it reads the variant.
Inline::Html(_) => {}
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn plain_text_flattens_markup_and_keeps_link_labels() {
let inline = vec![
Inline::Text("a ".into()),
Inline::Strong(vec![Inline::Text("b".into())]),
Inline::Text(" ".into()),
Inline::Link {
href: "/x".into(),
title: None,
label: vec![Inline::Emph(vec![Inline::Text("c".into())])],
},
Inline::SoftBreak,
Inline::Code("d".into()),
Inline::Html("<br>".into()),
];
assert_eq!(Doc::plain_text(&inline), "a b c d");
}
}

View file

@ -7,8 +7,11 @@
pub mod cache;
pub mod config;
pub mod error;
pub mod ir;
pub mod mime;
pub mod parse;
pub mod path;
pub mod preprocess;
pub mod site;
pub mod siteset;

655
core/src/parse.rs Normal file
View file

@ -0,0 +1,655 @@
//! Lowering Markdown into [`Doc`].
//!
//! `pulldown-cmark` is a pull parser whose events borrow from the source, which
//! maps directly onto the flat block sequence the formats want. The alternatives
//! build an arena AST that would only be flattened again.
//!
//! Only tables and strikethrough are enabled beyond CommonMark, matching what the
//! Python parsers actually handle. Events belonging to extensions that are off
//! are therefore never produced.
use std::path::Path;
use pulldown_cmark::{Alignment, CodeBlockKind, Event, Options, Parser, Tag, TagEnd};
use crate::error::Error;
use crate::ir::{Align, Block, Doc, Inline};
use crate::preprocess::{self, Segment};
/// Read a source file, expand its directives, and parse the result.
pub fn document(source: &Path, root: &Path) -> Result<Doc, Error> {
Ok(from_segments(&preprocess::expand(source, root)?))
}
/// Assemble a document from preprocessed segments.
///
/// A card break or an art block becomes a block directly, never passing through
/// the Markdown parser. That is what keeps a `{.card}` directive from surfacing
/// as literal paragraph text, which is a bug smolweb works around by stripping
/// the line before one of its two parsers sees it.
pub fn from_segments(segments: &[Segment]) -> Doc {
let mut doc = Doc { blocks: Vec::new(), first_h1: None };
for segment in segments {
match segment {
Segment::Markdown(text) => {
let mut parsed = markdown(text);
if doc.first_h1.is_none() {
doc.first_h1 = parsed.first_h1.take();
}
doc.blocks.append(&mut parsed.blocks);
}
Segment::Art { block_type, name, align, lines } => doc.blocks.push(Block::Art {
block_type: block_type.clone(),
name: name.clone(),
align: *align,
lines: lines.clone(),
}),
Segment::CardBreak { title } => {
doc.blocks.push(Block::CardBreak { title: title.clone() })
}
}
}
doc
}
/// Parse one run of Markdown.
pub fn markdown(text: &str) -> Doc {
let options = Options::ENABLE_TABLES | Options::ENABLE_STRIKETHROUGH;
let mut builder = Builder::default();
for event in Parser::new_ext(text, options) {
builder.handle(event);
}
builder.finish()
}
/// Data a tag needs when it closes, for the tags whose `TagEnd` carries none.
enum Pending {
Heading(u8),
Link { href: String, title: Option<String> },
Image { src: String, title: Option<String> },
}
struct ListFrame {
ordered: bool,
start: u64,
items: Vec<Vec<Block>>,
}
struct TableFrame {
alignments: Vec<Option<Align>>,
head: Vec<Vec<Inline>>,
rows: Vec<Vec<Vec<Inline>>>,
/// The row being filled, whether it belongs to the head or the body.
row: Vec<Vec<Inline>>,
in_head: bool,
}
struct CodeFrame {
info: Option<String>,
text: String,
}
struct Builder {
/// Stack of open block containers; index 0 is the document body.
blocks: Vec<Vec<Block>>,
/// Stack of open inline containers, for nested emphasis and link labels.
inlines: Vec<Vec<Inline>>,
pending: Vec<Pending>,
lists: Vec<ListFrame>,
tables: Vec<TableFrame>,
code: Option<CodeFrame>,
/// Accumulating the text of an open HTML block.
html: Option<String>,
/// Whether the innermost inline container was opened implicitly. A tight
/// list item holds its text with no `Paragraph` around it, so inline events
/// arrive with nothing open; without this they would be dropped.
implicit: bool,
first_h1: Option<String>,
}
impl Default for Builder {
fn default() -> Self {
Builder {
blocks: vec![Vec::new()],
inlines: Vec::new(),
pending: Vec::new(),
lists: Vec::new(),
tables: Vec::new(),
code: None,
html: None,
implicit: false,
first_h1: None,
}
}
}
impl Builder {
fn finish(mut self) -> Doc {
self.flush_implicit();
// Start and end events are balanced, so only the document frame is left.
let blocks = self.blocks.pop().unwrap_or_default();
Doc { blocks, first_h1: self.first_h1 }
}
fn handle(&mut self, event: Event<'_>) {
match event {
Event::Start(tag) => self.start(tag),
Event::End(end) => self.end(end),
Event::Text(text) => match &mut self.code {
Some(frame) => frame.text.push_str(&text),
None => self.push_inline(Inline::Text(text.to_string())),
},
Event::Code(code) => self.push_inline(Inline::Code(code.to_string())),
Event::SoftBreak => self.push_inline(Inline::SoftBreak),
Event::HardBreak => self.push_inline(Inline::HardBreak),
Event::Rule => self.push_block(Block::Rule),
Event::Html(html) => match &mut self.html {
Some(open) => open.push_str(&html),
None => self.push_block(Block::Html(html.to_string())),
},
Event::InlineHtml(html) => self.push_inline(Inline::Html(html.to_string())),
// Math, footnotes and task markers need extensions that are off.
_ => {}
}
}
fn start(&mut self, tag: Tag<'_>) {
// An implicit paragraph ends where the next block begins, so it has to
// be closed before that block is added or the order would invert.
if matches!(
tag,
Tag::Paragraph
| Tag::BlockQuote(_)
| Tag::CodeBlock(_)
| Tag::HtmlBlock
| Tag::List(_)
| Tag::Item
| Tag::Table(_)
) {
self.flush_implicit();
}
match tag {
Tag::Paragraph => self.open_inlines(),
Tag::Heading { level, .. } => {
self.pending.push(Pending::Heading(level as u8));
self.open_inlines();
}
Tag::BlockQuote(_) => self.blocks.push(Vec::new()),
Tag::CodeBlock(kind) => {
let info = match kind {
CodeBlockKind::Fenced(info) if !info.is_empty() => Some(info.to_string()),
_ => None,
};
self.code = Some(CodeFrame { info, text: String::new() });
}
Tag::HtmlBlock => self.html = Some(String::new()),
Tag::List(start) => self.lists.push(ListFrame {
ordered: start.is_some(),
start: start.unwrap_or(1),
items: Vec::new(),
}),
Tag::Item => self.blocks.push(Vec::new()),
Tag::Table(alignments) => self.tables.push(TableFrame {
alignments: alignments.into_iter().map(align_of).collect(),
head: Vec::new(),
rows: Vec::new(),
row: Vec::new(),
in_head: false,
}),
Tag::TableHead => {
if let Some(table) = self.tables.last_mut() {
table.in_head = true;
table.row.clear();
}
}
Tag::TableRow => {
if let Some(table) = self.tables.last_mut() {
table.row.clear();
}
}
Tag::TableCell => self.open_inlines(),
Tag::Emphasis | Tag::Strong | Tag::Strikethrough => self.open_inlines(),
Tag::Link { dest_url, title, .. } => {
self.pending
.push(Pending::Link { href: dest_url.to_string(), title: non_empty(&title) });
self.open_inlines();
}
Tag::Image { dest_url, title, .. } => {
self.pending
.push(Pending::Image { src: dest_url.to_string(), title: non_empty(&title) });
self.open_inlines();
}
_ => {}
}
}
fn end(&mut self, end: TagEnd) {
match end {
TagEnd::Paragraph => {
let inline = self.close_inlines();
self.push_block(Block::Paragraph(inline));
}
TagEnd::Heading(_) => {
let inline = self.close_inlines();
let level = match self.pending.pop() {
Some(Pending::Heading(level)) => level,
other => {
if let Some(value) = other {
self.pending.push(value);
}
1
}
};
if level == 1 && self.first_h1.is_none() {
self.first_h1 = Some(Doc::plain_text(&inline));
}
self.push_block(Block::Heading { level, inline });
}
TagEnd::BlockQuote(_) => {
self.flush_implicit();
let blocks = self.blocks.pop().unwrap_or_default();
self.push_block(Block::BlockQuote(blocks));
}
TagEnd::CodeBlock => {
if let Some(frame) = self.code.take() {
let lines = frame.text.lines().map(str::to_string).collect();
self.push_block(Block::CodeBlock { info: frame.info, lines });
}
}
TagEnd::HtmlBlock => {
if let Some(html) = self.html.take() {
self.push_block(Block::Html(html));
}
}
TagEnd::List(_) => {
self.flush_implicit();
if let Some(frame) = self.lists.pop() {
self.push_block(Block::List {
ordered: frame.ordered,
start: frame.start,
items: frame.items,
});
}
}
TagEnd::Item => {
self.flush_implicit();
let item = self.blocks.pop().unwrap_or_default();
if let Some(list) = self.lists.last_mut() {
list.items.push(item);
}
}
TagEnd::Table => {
if let Some(frame) = self.tables.pop() {
self.push_block(Block::Table {
alignments: frame.alignments,
head: frame.head,
rows: frame.rows,
});
}
}
TagEnd::TableHead => {
if let Some(table) = self.tables.last_mut() {
table.head = std::mem::take(&mut table.row);
table.in_head = false;
}
}
TagEnd::TableRow => {
if let Some(table) = self.tables.last_mut() {
let row = std::mem::take(&mut table.row);
table.rows.push(row);
}
}
TagEnd::TableCell => {
let cell = self.close_inlines();
if let Some(table) = self.tables.last_mut() {
table.row.push(cell);
}
}
TagEnd::Emphasis => {
let inner = self.close_inlines();
self.push_inline(Inline::Emph(inner));
}
TagEnd::Strong => {
let inner = self.close_inlines();
self.push_inline(Inline::Strong(inner));
}
TagEnd::Strikethrough => {
let inner = self.close_inlines();
self.push_inline(Inline::Strike(inner));
}
TagEnd::Link => {
let label = self.close_inlines();
if let Some(Pending::Link { href, title }) = self.pending.pop() {
self.push_inline(Inline::Link { href, title, label });
}
}
TagEnd::Image => {
let alt = self.close_inlines();
if let Some(Pending::Image { src, title }) = self.pending.pop() {
self.push_inline(Inline::Image { src, title, alt });
}
}
_ => {}
}
}
fn open_inlines(&mut self) {
self.inlines.push(Vec::new());
}
fn close_inlines(&mut self) -> Vec<Inline> {
self.inlines.pop().unwrap_or_default()
}
/// Add an inline to the innermost open inline container, opening an implicit
/// paragraph if nothing is open. That happens for every tight list item,
/// whose text pulldown-cmark emits with no `Paragraph` around it.
fn push_inline(&mut self, inline: Inline) {
if self.inlines.is_empty() {
self.inlines.push(Vec::new());
self.implicit = true;
}
if let Some(frame) = self.inlines.last_mut() {
frame.push(inline);
}
}
/// Close an implicit paragraph, if one is open, into the current block
/// container. Pushes directly rather than through `push_block`, which calls
/// this.
fn flush_implicit(&mut self) {
if !self.implicit {
return;
}
self.implicit = false;
let inline = self.inlines.pop().unwrap_or_default();
if !inline.is_empty()
&& let Some(frame) = self.blocks.last_mut()
{
frame.push(Block::Paragraph(inline));
}
}
fn push_block(&mut self, block: Block) {
self.flush_implicit();
if let Some(frame) = self.blocks.last_mut() {
frame.push(block);
}
}
}
fn align_of(alignment: Alignment) -> Option<Align> {
match alignment {
Alignment::None => None,
Alignment::Left => Some(Align::Left),
Alignment::Center => Some(Align::Center),
Alignment::Right => Some(Align::Right),
}
}
fn non_empty(value: &str) -> Option<String> {
(!value.is_empty()).then(|| value.to_string())
}
#[cfg(test)]
mod tests {
use super::*;
fn blocks(text: &str) -> Vec<Block> {
markdown(text).blocks
}
fn text(value: &str) -> Vec<Inline> {
vec![Inline::Text(value.into())]
}
#[test]
fn parses_headings_and_paragraphs() {
assert_eq!(
blocks("# Title\n\nBody.\n"),
vec![
Block::Heading { level: 1, inline: text("Title") },
Block::Paragraph(text("Body.")),
]
);
}
#[test]
fn parses_setext_headings() {
// wapdown's parser handles these and md2txt's does not, so unifying the
// two parsers gains them for the text formats.
assert_eq!(
blocks("Title\n=====\n"),
vec![Block::Heading { level: 1, inline: text("Title") }]
);
}
#[test]
fn records_the_first_level_one_heading() {
let doc = markdown("## Second\n\n# First\n\n# Another\n");
assert_eq!(doc.first_h1.as_deref(), Some("First"));
// A document with no h1 has nothing to derive a title from.
assert_eq!(markdown("## Only\n").first_h1, None);
}
#[test]
fn parses_inline_markup() {
let Block::Paragraph(inline) = &blocks("a *b* **c** ~~d~~ `e`\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(
inline,
&vec![
Inline::Text("a ".into()),
Inline::Emph(text("b")),
Inline::Text(" ".into()),
Inline::Strong(text("c")),
Inline::Text(" ".into()),
Inline::Strike(text("d")),
Inline::Text(" ".into()),
Inline::Code("e".into()),
]
);
}
#[test]
fn parses_links_and_images_with_their_labels() {
let Block::Paragraph(inline) = &blocks("[a *b*](/x \"T\") ![alt](/i.png)\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(
inline[0],
Inline::Link {
href: "/x".into(),
title: Some("T".into()),
label: vec![Inline::Text("a ".into()), Inline::Emph(text("b"))],
}
);
assert_eq!(
inline[2],
Inline::Image { src: "/i.png".into(), title: None, alt: text("alt") }
);
}
#[test]
fn parses_nested_lists() {
let parsed = blocks("- a\n - b\n- c\n");
let Block::List { ordered, start, items } = &parsed[0] else { panic!("expected a list") };
assert!(!ordered);
assert_eq!(*start, 1);
assert_eq!(items.len(), 2);
// The first item holds its own paragraph and the nested list.
assert_eq!(items[0][0], Block::Paragraph(text("a")));
assert!(matches!(items[0][1], Block::List { .. }));
assert_eq!(items[1][0], Block::Paragraph(text("c")));
}
#[test]
fn a_tight_list_items_text_is_kept() {
// Regression: pulldown-cmark emits a tight item's text with no Paragraph
// around it, so an implicit one has to be opened or the text is dropped.
let Block::List { items, .. } = &blocks("- a\n- b\n")[0] else { panic!("expected a list") };
assert_eq!(
items,
&vec![vec![Block::Paragraph(text("a"))], vec![Block::Paragraph(text("b"))]]
);
}
#[test]
fn a_tight_item_keeps_its_inline_markup_in_order() {
let Block::List { items, .. } = &blocks("- a *b* c\n")[0] else {
panic!("expected a list")
};
assert_eq!(
items[0],
vec![Block::Paragraph(vec![
Inline::Text("a ".into()),
Inline::Emph(text("b")),
Inline::Text(" c".into()),
])]
);
}
#[test]
fn a_tight_item_opening_with_markup_is_still_one_paragraph() {
let Block::List { items, .. } = &blocks("- *a* b\n")[0] else { panic!("expected a list") };
assert_eq!(
items[0],
vec![Block::Paragraph(vec![Inline::Emph(text("a")), Inline::Text(" b".into()),])]
);
}
#[test]
fn a_loose_list_is_shaped_the_same_as_a_tight_one() {
// The distinction is pulldown-cmark's, not ours: both yield a paragraph
// per item, so no format has to know which it was.
let Block::List { items: tight, .. } = &blocks("- a\n- b\n")[0] else { panic!("list") };
let Block::List { items: loose, .. } = &blocks("- a\n\n- b\n")[0] else { panic!("list") };
assert_eq!(tight, loose);
}
#[test]
fn an_ordered_list_keeps_its_starting_number() {
let Block::List { ordered, start, items } = &blocks("3. a\n4. b\n")[0] else {
panic!("expected a list")
};
assert!(ordered);
assert_eq!(*start, 3);
assert_eq!(items.len(), 2);
}
#[test]
fn parses_nested_blockquotes() {
// Depth comes from the nesting, so there is one place it is recorded.
let parsed = blocks("> outer\n>\n> > inner\n");
let Block::BlockQuote(outer) = &parsed[0] else { panic!("expected a quote") };
assert_eq!(outer[0], Block::Paragraph(text("outer")));
assert_eq!(outer[1], Block::BlockQuote(vec![Block::Paragraph(text("inner"))]));
}
#[test]
fn parses_fenced_and_indented_code() {
assert_eq!(
blocks("```rust\nlet x = 1;\nlet y = 2;\n```\n"),
vec![Block::CodeBlock {
info: Some("rust".into()),
lines: vec!["let x = 1;".into(), "let y = 2;".into()],
}]
);
assert_eq!(
blocks(" indented\n"),
vec![Block::CodeBlock { info: None, lines: vec!["indented".into()] }]
);
}
#[test]
fn parses_tables_with_their_alignments() {
let parsed = blocks("| a | b |\n| :-- | --: |\n| 1 | 2 |\n| 3 | 4 |\n");
let Block::Table { alignments, head, rows } = &parsed[0] else {
panic!("expected a table")
};
assert_eq!(alignments, &vec![Some(Align::Left), Some(Align::Right)]);
assert_eq!(head, &vec![text("a"), text("b")]);
assert_eq!(rows, &vec![vec![text("1"), text("2")], vec![text("3"), text("4")]]);
}
#[test]
fn parses_rules_and_breaks() {
assert_eq!(blocks("---\n"), vec![Block::Rule]);
let Block::Paragraph(inline) = &blocks("a\nb\n")[0] else { panic!("expected a paragraph") };
assert_eq!(inline[1], Inline::SoftBreak);
let Block::Paragraph(inline) = &blocks("a \nb\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(inline[1], Inline::HardBreak);
}
#[test]
fn keeps_raw_html_rather_than_dropping_it() {
// XHTML-MP passes it through and the text formats strip it; the parser
// does not get to decide.
let parsed = blocks("<div>x</div>\n");
assert!(matches!(&parsed[0], Block::Html(html) if html.contains("<div>")));
let Block::Paragraph(inline) = &blocks("a <b>c</b>\n")[0] else {
panic!("expected a paragraph")
};
assert_eq!(inline[1], Inline::Html("<b>".into()));
}
// -- Segment assembly --------------------------------------------------
#[test]
fn card_breaks_and_art_become_blocks_without_passing_through_the_parser() {
let segments = vec![
Segment::Markdown("Intro.\n".into()),
Segment::CardBreak { title: Some("Weather".into()) },
Segment::Markdown("Cold.\n".into()),
Segment::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec!["/\\".into()],
},
];
let doc = from_segments(&segments);
assert_eq!(
doc.blocks,
vec![
Block::Paragraph(text("Intro.")),
Block::CardBreak { title: Some("Weather".into()) },
Block::Paragraph(text("Cold.")),
Block::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec!["/\\".into()],
},
]
);
}
#[test]
fn a_card_directive_never_appears_as_text() {
// The leak smolweb works around by stripping the line before md2txt sees
// it: here the directive cannot reach the Markdown parser at all.
let dir = tempfile::tempdir().unwrap();
let root = dir.path().canonicalize().unwrap();
std::fs::write(root.join("page.md"), "Intro.\n\n{.card Weather}\nCold.\n").unwrap();
let doc = document(&root.join("page.md"), &root).unwrap();
let rendered = format!("{:?}", doc.blocks);
assert!(!rendered.contains("{.card"), "{rendered}");
assert!(doc.blocks.contains(&Block::CardBreak { title: Some("Weather".into()) }));
}
#[test]
fn the_title_comes_from_the_first_markdown_run_that_has_one() {
let segments = vec![
Segment::CardBreak { title: None },
Segment::Markdown("## not it\n".into()),
Segment::Markdown("# Found\n".into()),
Segment::Markdown("# Later\n".into()),
];
assert_eq!(from_segments(&segments).first_h1.as_deref(), Some("Found"));
}
}

537
core/src/preprocess.rs Normal file
View file

@ -0,0 +1,537 @@
//! The line-level pass that runs before Markdown parsing.
//!
//! Four directives live outside CommonMark, and all four occupy a whole line, so
//! they are recognised here rather than by extending the Markdown parser. That
//! is also how the Python does it; the difference is that this pass emits typed
//! segments instead of encoding them as sentinel strings inside the line stream.
//!
//! This module is the only part of the rendering pipeline that opens files.
//! Keeping it so means the root-containment check has exactly one home, which is
//! what closes the traversal smolweb has: md2txt resolves an include target and
//! checks only that it exists, so `{.include ../../../../etc/passwd}` in any
//! served document reads and emits that file.
use std::collections::BTreeSet;
use std::fs;
use std::path::{Path, PathBuf};
use crate::error::{Error, IncludeReason};
use crate::ir::Align;
/// How deep includes may nest. The cycle set alone does not bound a long chain
/// that never repeats a file.
const MAX_DEPTH: usize = 16;
/// Caps on the expanded result. The cycle set is per-*stack*, so a diamond —
/// `a` includes `b` and `c`, both include `d` — fans out exponentially without
/// ever repeating a file on one path. smolweb has nothing that stops this.
const MAX_LINES: usize = 200_000;
const MAX_BYTES: usize = 8 * 1024 * 1024;
/// A run of input, classified. Markdown runs are parsed; the rest become blocks
/// directly, so no directive can ever be mistaken for paragraph text.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Segment {
Markdown(String),
Art { block_type: String, name: String, align: Option<Align>, lines: Vec<String> },
CardBreak { title: Option<String> },
}
/// Read `source` and expand it into segments.
///
/// `root` bounds every path this may read: an include or art target that
/// canonicalises outside it is refused, not followed.
pub fn expand(source: &Path, root: &Path) -> Result<Vec<Segment>, Error> {
let mut out = Segments::default();
let mut stack = BTreeSet::new();
let canonical = canonical_within(source, root)?;
stack.insert(canonical.clone());
expand_file(&canonical, root, &mut stack, 0, &mut out)?;
Ok(out.finish())
}
/// Accumulates segments, merging consecutive Markdown lines into one run so the
/// Markdown parser sees whole constructs rather than line fragments.
#[derive(Default)]
struct Segments {
done: Vec<Segment>,
markdown: String,
lines: usize,
bytes: usize,
}
impl Segments {
fn push_line(&mut self, line: &str) -> Result<(), Error> {
self.lines += 1;
self.bytes += line.len();
if self.lines > MAX_LINES || self.bytes > MAX_BYTES {
return Err(Error::Include { path: PathBuf::new(), reason: IncludeReason::TooLarge });
}
self.markdown.push_str(line);
self.markdown.push('\n');
Ok(())
}
fn push(&mut self, segment: Segment) {
self.flush();
self.done.push(segment);
}
fn flush(&mut self) {
if !self.markdown.is_empty() {
self.done.push(Segment::Markdown(std::mem::take(&mut self.markdown)));
}
}
fn finish(mut self) -> Vec<Segment> {
self.flush();
self.done
}
}
fn expand_file(
path: &Path,
root: &Path,
stack: &mut BTreeSet<PathBuf>,
depth: usize,
out: &mut Segments,
) -> Result<(), Error> {
let text =
fs::read_to_string(path).map_err(|cause| Error::Io { path: path.to_path_buf(), cause })?;
let base = path.parent().unwrap_or(root);
for line in text.lines() {
if let Some(title) = card_break(line) {
out.push(Segment::CardBreak { title: title.map(str::to_string) });
continue;
}
if let Some(pieces) = art_pieces(line) {
for (label, target) in pieces {
let (block_type, name, align) = parse_label(label);
let file = resolve(target, base, root)?;
let art = fs::read_to_string(&file)
.map_err(|cause| Error::Io { path: file.clone(), cause })?;
out.push(Segment::Art {
block_type,
name,
align,
lines: art.lines().map(str::to_string).collect(),
});
}
continue;
}
if let Some(target) = include_target(line) {
if depth + 1 > MAX_DEPTH {
return Err(Error::Include {
path: path.to_path_buf(),
reason: IncludeReason::TooDeep,
});
}
let file = resolve(target, base, root)?;
if !stack.insert(file.clone()) {
return Err(Error::Include { path: file, reason: IncludeReason::Circular });
}
expand_file(&file, root, stack, depth + 1, out)?;
stack.remove(&file);
continue;
}
out.push_line(line)?;
}
Ok(())
}
/// Resolve a directive's target relative to the including file, and require the
/// result to be a readable file inside the content root.
fn resolve(target: &str, base: &Path, root: &Path) -> Result<PathBuf, Error> {
canonical_within(&base.join(target), root)
}
fn canonical_within(path: &Path, root: &Path) -> Result<PathBuf, Error> {
let canonical = path
.canonicalize()
.map_err(|_| Error::Include { path: path.to_path_buf(), reason: IncludeReason::Missing })?;
if !canonical.starts_with(root) || !canonical.is_file() {
return Err(Error::Include { path: canonical, reason: IncludeReason::Outside });
}
Ok(canonical)
}
// -- Directive recognition ------------------------------------------------
//
// Hand-written rather than regex-driven: each pattern is an anchored prefix and
// suffix around one capture, and a dependency for five of those is not a trade
// worth making in a codebase meant to stay auditable.
/// `{.card}` or `{.card Title}` alone on a line. The outer `Option` is whether
/// the line is a card break at all; the inner one is whether it carries a title.
pub fn card_break(line: &str) -> Option<Option<&str>> {
let rest = line.trim().strip_prefix('{')?.trim_start().strip_prefix(".card")?;
let inner = rest.trim_end().strip_suffix('}')?;
if inner.is_empty() {
return Some(None);
}
// A following character must be whitespace, or this is `{.cardsomething}`.
if !inner.starts_with(char::is_whitespace) {
return None;
}
let title = inner.trim();
Some((!title.is_empty()).then_some(title))
}
/// `![[target]]` or `{.include target}` alone on a line.
pub fn include_target(line: &str) -> Option<&str> {
let trimmed = line.trim();
if let Some(inner) = trimmed.strip_prefix("![[").and_then(|r| r.strip_suffix("]]")) {
return unquote(inner);
}
let rest = trimmed.strip_prefix('{')?.trim_start().strip_prefix(".include")?;
if !rest.starts_with(char::is_whitespace) {
return None;
}
unquote(rest.trim_end().strip_suffix('}')?)
}
/// Strip surrounding matched quotes, as the Python's target normalisation does.
fn unquote(value: &str) -> Option<&str> {
let trimmed = value.trim();
let inner = match (trimmed.chars().next(), trimmed.chars().last()) {
(Some(first @ ('\'' | '"')), Some(last)) if first == last && trimmed.len() >= 2 => {
trimmed[1..trimmed.len() - 1].trim()
}
_ => trimmed,
};
(!inner.is_empty()).then_some(inner)
}
/// `#[label](target)` pieces filling a whole line, as `(label, target)` pairs.
///
/// One piece is the block form, several the inline form; the Python accepts the
/// inline form only when nothing but whitespace surrounds the pieces, which is
/// what makes this a block-level directive and keeps it from splitting a
/// sentence.
pub fn art_pieces(line: &str) -> Option<Vec<(&str, &str)>> {
let mut rest = line.trim();
// The block form allows a trailing MultiMarkdown attribute list. It is
// recognised so the line is still treated as art, but no format reads it.
if let Some(open) = rest.rfind("{:")
&& rest.ends_with('}')
{
rest = rest[..open].trim_end();
}
let mut pieces = Vec::new();
while !rest.is_empty() {
let (label, target, consumed) = one_art(rest)?;
pieces.push((label, target));
rest = rest[consumed..].trim_start();
}
(!pieces.is_empty()).then_some(pieces)
}
/// One `#[label](target)` at the start of `text`, with the length it consumed.
fn one_art(text: &str) -> Option<(&str, &str, usize)> {
let after_open = text.strip_prefix("#[")?;
let label_end = after_open.find(']')?;
let label = &after_open[..label_end];
let after_label = &after_open[label_end + 1..];
let after_paren = after_label.strip_prefix('(')?;
let target_end = after_paren.find(')')?;
let target = &after_paren[..target_end];
if label.is_empty() || target.is_empty() {
return None;
}
// "#[" + label + "](" + target + ")"
let consumed = 2 + label.len() + 2 + target.len() + 1;
Some((label, target, consumed))
}
/// Split an art label into its block type, name and alignment.
///
/// Tokens beginning with a colon set the alignment; the first remaining token is
/// the block type and the rest its name. An absent type is `custom`, matching
/// the Python.
fn parse_label(label: &str) -> (String, String, Option<Align>) {
let mut align = None;
let mut words = Vec::new();
for token in label.split_whitespace() {
match token.strip_prefix(':') {
Some(tag) => {
align = match tag.to_ascii_lowercase().as_str() {
"left" => Some(Align::Left),
"right" => Some(Align::Right),
"center" | "centre" => Some(Align::Center),
// An unrecognised tag is ignored rather than treated as a name.
_ => align,
};
}
None => words.push(token),
}
}
let block_type = words.first().copied().unwrap_or("custom").to_string();
let name = words.get(1..).map(|rest| rest.join(" ")).unwrap_or_default();
(block_type, name, align)
}
#[cfg(test)]
mod tests {
use super::*;
// -- Directive recognition ---------------------------------------------
#[test]
fn recognises_card_breaks() {
assert_eq!(card_break("{.card}"), Some(None));
assert_eq!(card_break(" {.card} "), Some(None));
assert_eq!(card_break("{ .card }"), Some(None));
assert_eq!(card_break("{.card Weather}"), Some(Some("Weather")));
assert_eq!(card_break("{ .card Two Words }"), Some(Some("Two Words")));
}
#[test]
fn rejects_near_misses_for_card_breaks() {
assert_eq!(card_break("{.cardinal}"), None);
assert_eq!(card_break("text {.card}"), None);
assert_eq!(card_break("{.card"), None);
assert_eq!(card_break("{.cards Weather}"), None);
}
#[test]
fn recognises_includes_in_both_spellings() {
assert_eq!(include_target("![[notes.md]]"), Some("notes.md"));
assert_eq!(include_target(" ![[a/b.md]] "), Some("a/b.md"));
assert_eq!(include_target("{.include notes.md}"), Some("notes.md"));
assert_eq!(include_target("{ .include notes.md }"), Some("notes.md"));
// Quoted targets, as the Python's normalisation allows.
assert_eq!(include_target("{.include \"a b.md\"}"), Some("a b.md"));
assert_eq!(include_target("![['q.md']]"), Some("q.md"));
}
#[test]
fn rejects_near_misses_for_includes() {
assert_eq!(include_target("see ![[notes.md]] there"), None);
assert_eq!(include_target("{.includes notes.md}"), None);
assert_eq!(include_target("{.include}"), None);
assert_eq!(include_target("![[]]"), None);
}
#[test]
fn recognises_art_in_block_and_inline_forms() {
assert_eq!(art_pieces("#[banner](logo.txt)"), Some(vec![("banner", "logo.txt")]));
assert_eq!(
art_pieces("#[a](one.txt) #[b](two.txt)"),
Some(vec![("a", "one.txt"), ("b", "two.txt")])
);
// A trailing attribute list keeps the line recognised as art.
assert_eq!(art_pieces("#[banner](logo.txt){: .wide}"), Some(vec![("banner", "logo.txt")]));
}
#[test]
fn art_must_fill_the_whole_line() {
// Otherwise it would split a sentence, which is why the Python requires
// the surrounding text to be whitespace only.
assert_eq!(art_pieces("see #[a](one.txt)"), None);
assert_eq!(art_pieces("#[a](one.txt) trailing"), None);
assert_eq!(art_pieces("plain text"), None);
assert_eq!(art_pieces("#[a]()"), None);
}
#[test]
fn parses_art_labels() {
assert_eq!(parse_label("banner"), ("banner".into(), String::new(), None));
assert_eq!(parse_label("figlet Big Title"), ("figlet".into(), "Big Title".into(), None));
assert_eq!(
parse_label(":center banner Logo"),
("banner".into(), "Logo".into(), Some(Align::Center))
);
// The British spelling is accepted and folded onto the same value.
assert_eq!(parse_label(":centre x").2, Some(Align::Center));
assert_eq!(parse_label(":LEFT x").2, Some(Align::Left));
// No type at all falls back to `custom`, as the Python does.
assert_eq!(parse_label(":right"), ("custom".into(), String::new(), Some(Align::Right)));
// An unknown colon tag is dropped, not taken as the type.
assert_eq!(parse_label(":nonsense banner"), ("banner".into(), String::new(), None));
}
// -- Expansion ---------------------------------------------------------
struct Tree(tempfile::TempDir);
impl Tree {
fn new() -> Self {
Tree(tempfile::tempdir().unwrap())
}
fn root(&self) -> PathBuf {
self.0.path().canonicalize().unwrap()
}
fn write(&self, rel: &str, body: &str) -> PathBuf {
let path = self.0.path().join(rel);
if let Some(parent) = path.parent() {
fs::create_dir_all(parent).unwrap();
}
fs::write(&path, body).unwrap();
path
}
fn expand(&self, rel: &str) -> Result<Vec<Segment>, Error> {
expand(&self.0.path().join(rel), &self.root())
}
}
#[test]
fn a_file_with_no_directives_is_one_markdown_run() {
let tree = Tree::new();
tree.write("page.md", "# Title\n\nBody.\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Markdown("# Title\n\nBody.\n".into())]
);
}
#[test]
fn card_breaks_split_the_markdown_runs() {
let tree = Tree::new();
tree.write("page.md", "Intro.\n\n{.card Weather}\nCold.\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![
Segment::Markdown("Intro.\n\n".into()),
Segment::CardBreak { title: Some("Weather".into()) },
Segment::Markdown("Cold.\n".into()),
]
);
}
#[test]
fn an_include_splices_the_targets_lines() {
let tree = Tree::new();
tree.write("part.md", "Shared text.\n");
tree.write("page.md", "Before.\n\n{.include part.md}\n\nAfter.\n");
// The spliced lines join the surrounding run: an include is a text
// substitution, not a structural boundary.
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Markdown("Before.\n\nShared text.\n\nAfter.\n".into())]
);
}
#[test]
fn a_wikilink_include_resolves_relative_to_the_including_file() {
let tree = Tree::new();
tree.write("sub/part.md", "Nested shared.\n");
tree.write("sub/page.md", "![[part.md]]\n");
assert_eq!(
tree.expand("sub/page.md").unwrap(),
vec![Segment::Markdown("Nested shared.\n".into())]
);
}
#[test]
fn art_is_read_into_the_segment() {
let tree = Tree::new();
tree.write("logo.txt", " /\\ \n/__\\\n");
tree.write("page.md", "#[:center banner Logo](logo.txt)\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec![" /\\ ".into(), "/__\\".into()],
}]
);
}
#[test]
fn an_include_outside_the_root_is_refused() {
// The traversal smolweb has: md2txt resolves the target and checks only
// that it exists, with no containment check at all.
let tree = Tree::new();
let outside = tree.0.path().parent().unwrap().join("itsybitsy-escape.md");
fs::write(&outside, "secret\n").unwrap();
tree.write("page.md", "{.include ../itsybitsy-escape.md}\n");
let err = tree.expand("page.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Outside, .. }), "{err}");
fs::remove_file(outside).unwrap();
}
#[test]
fn an_art_target_outside_the_root_is_refused() {
let tree = Tree::new();
let outside = tree.0.path().parent().unwrap().join("itsybitsy-art-escape.txt");
fs::write(&outside, "secret\n").unwrap();
tree.write("page.md", "#[banner x](../itsybitsy-art-escape.txt)\n");
let err = tree.expand("page.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Outside, .. }), "{err}");
fs::remove_file(outside).unwrap();
}
#[test]
fn a_missing_target_is_reported_not_ignored() {
let tree = Tree::new();
tree.write("page.md", "{.include absent.md}\n");
let err = tree.expand("page.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Missing, .. }), "{err}");
}
#[test]
fn a_circular_include_is_refused() {
let tree = Tree::new();
tree.write("a.md", "A\n{.include b.md}\n");
tree.write("b.md", "B\n{.include a.md}\n");
let err = tree.expand("a.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Circular, .. }), "{err}");
}
#[test]
fn a_self_include_is_refused() {
let tree = Tree::new();
tree.write("a.md", "{.include a.md}\n");
let err = tree.expand("a.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Circular, .. }), "{err}");
}
#[test]
fn a_chain_deeper_than_the_cap_is_refused() {
// No file repeats, so the cycle set never fires: this is what the depth
// cap is for.
let tree = Tree::new();
for i in 0..=MAX_DEPTH + 1 {
tree.write(&format!("n{i}.md"), &format!("line {i}\n{{.include n{}.md}}\n", i + 1));
}
let err = tree.expand("n0.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::TooDeep, .. }), "{err}");
}
#[test]
fn a_diamond_fan_out_is_stopped_by_the_size_cap() {
// Each level doubles and no file repeats on any single path, so neither
// the cycle set nor the depth cap catches it. smolweb expands this until
// it runs out of memory.
let tree = Tree::new();
tree.write("leaf.md", &"filler line\n".repeat(64));
let mut previous = "leaf.md".to_string();
for level in 0..MAX_DEPTH - 2 {
let name = format!("l{level}.md");
tree.write(&name, &format!("{{.include {previous}}}\n{{.include {previous}}}\n"));
previous = name;
}
let err = tree.expand(&previous).unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::TooLarge, .. }), "{err}");
}
#[test]
fn a_repeated_include_on_separate_paths_is_allowed() {
// The cycle set is per-stack, so including one shared file twice in
// sequence is fine and must not be mistaken for a cycle.
let tree = Tree::new();
tree.write("part.md", "shared\n");
tree.write("page.md", "{.include part.md}\n{.include part.md}\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Markdown("shared\nshared\n".into())]
);
}
}