feat: parse markdown and custom directives into one representation
This commit is contained in:
parent
3031b8acdd
commit
e5583bbd89
8 changed files with 1410 additions and 2 deletions
|
|
@ -8,6 +8,7 @@ license = "Apache-2.0"
|
|||
publish = false
|
||||
|
||||
[dependencies]
|
||||
pulldown-cmark = { version = "0.13", default-features = false }
|
||||
serde = { version = "1.0", features = ["derive"] }
|
||||
toml = "1.1"
|
||||
|
||||
|
|
|
|||
|
|
@ -25,6 +25,37 @@ pub enum Error {
|
|||
/// the registry's full id set, which is what tells an operator whether the
|
||||
/// name is a typo or a missing cargo feature.
|
||||
UnknownFormat { listener: String, format: String, available: Vec<String> },
|
||||
|
||||
/// An include or ASCII-art target cannot be used. `path` is where the
|
||||
/// directive actually pointed, resolved, which is the thing an author needs
|
||||
/// to see when a relative target is wrong.
|
||||
Include { path: PathBuf, reason: IncludeReason },
|
||||
}
|
||||
|
||||
/// Why a directive's target was refused.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum IncludeReason {
|
||||
/// Resolved outside the content root, or is not a regular file.
|
||||
Outside,
|
||||
Missing,
|
||||
/// Already being expanded further up the stack.
|
||||
Circular,
|
||||
TooDeep,
|
||||
/// The expansion exceeded its line or byte budget.
|
||||
TooLarge,
|
||||
}
|
||||
|
||||
impl fmt::Display for IncludeReason {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
let text = match self {
|
||||
IncludeReason::Outside => "is outside the content root",
|
||||
IncludeReason::Missing => "does not exist",
|
||||
IncludeReason::Circular => "is already being included",
|
||||
IncludeReason::TooDeep => "includes nest too deeply",
|
||||
IncludeReason::TooLarge => "expands to more than the size limit",
|
||||
};
|
||||
f.write_str(text)
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Display for Error {
|
||||
|
|
@ -37,6 +68,9 @@ impl fmt::Display for Error {
|
|||
Error::DuplicateHost { host, first, second } => {
|
||||
write!(f, "host '{host}' is claimed by both site '{first}' and site '{second}'")
|
||||
}
|
||||
Error::Include { path, reason } => {
|
||||
write!(f, "include target {} {reason}", path.display())
|
||||
}
|
||||
Error::UnknownFormat { listener, format, available } => write!(
|
||||
f,
|
||||
"listener '{listener}' wants format '{format}', which this build does not \
|
||||
|
|
|
|||
142
core/src/ir.rs
Normal file
142
core/src/ir.rs
Normal file
|
|
@ -0,0 +1,142 @@
|
|||
//! The one parsed representation every output format consumes.
|
||||
//!
|
||||
//! smolweb parses each document twice, with two hand-rolled regex parsers that
|
||||
//! share seven identical patterns but disagree on the edges: that is where the
|
||||
//! `{.card}` directive leaks into gemtext as literal text, and why the two
|
||||
//! libraries carry two divergent sets of defaults. Parsing once into this
|
||||
//! structure removes the class of bug rather than the instances.
|
||||
//!
|
||||
//! It is a flat block sequence rather than a tree of nodes because that is what
|
||||
//! the consumers want: WML packs a linear run of blocks into byte-budgeted
|
||||
//! cards, and the text renderers are a fold over blocks.
|
||||
|
||||
/// A parsed document.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub struct Doc {
|
||||
pub blocks: Vec<Block>,
|
||||
/// The first level-1 heading as plain text, used to derive a page title when
|
||||
/// the directory config does not set one.
|
||||
pub first_h1: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Block {
|
||||
Paragraph(Vec<Inline>),
|
||||
Heading {
|
||||
level: u8,
|
||||
inline: Vec<Inline>,
|
||||
},
|
||||
/// `info` is the fence's info string, so a renderer can label or highlight.
|
||||
CodeBlock {
|
||||
info: Option<String>,
|
||||
lines: Vec<String>,
|
||||
},
|
||||
/// Nesting carries the depth, so there is one place it is recorded.
|
||||
BlockQuote(Vec<Block>),
|
||||
List {
|
||||
ordered: bool,
|
||||
start: u64,
|
||||
items: Vec<Vec<Block>>,
|
||||
},
|
||||
Rule,
|
||||
Table {
|
||||
alignments: Vec<Option<Align>>,
|
||||
head: Vec<Vec<Inline>>,
|
||||
rows: Vec<Vec<Vec<Inline>>>,
|
||||
},
|
||||
/// A verbatim block lifted from a file by `#[label](art.txt)`. Already read,
|
||||
/// because nothing downstream of the parser touches the filesystem.
|
||||
Art {
|
||||
block_type: String,
|
||||
name: String,
|
||||
align: Option<Align>,
|
||||
lines: Vec<String>,
|
||||
},
|
||||
/// A `{.card Title}` divider. Formats that paginate start a new screen here;
|
||||
/// every other format emits nothing at all for it.
|
||||
CardBreak {
|
||||
title: Option<String>,
|
||||
},
|
||||
/// Raw block HTML. Kept rather than dropped so XHTML-MP can pass it through
|
||||
/// and the text formats can strip it, instead of the parser deciding.
|
||||
Html(String),
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Inline {
|
||||
Text(String),
|
||||
Code(String),
|
||||
Emph(Vec<Inline>),
|
||||
Strong(Vec<Inline>),
|
||||
Strike(Vec<Inline>),
|
||||
Link {
|
||||
href: String,
|
||||
title: Option<String>,
|
||||
label: Vec<Inline>,
|
||||
},
|
||||
Image {
|
||||
src: String,
|
||||
title: Option<String>,
|
||||
alt: Vec<Inline>,
|
||||
},
|
||||
/// A line break the source left soft; renderers that wrap ignore it.
|
||||
SoftBreak,
|
||||
HardBreak,
|
||||
Html(String),
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum Align {
|
||||
Left,
|
||||
Center,
|
||||
Right,
|
||||
}
|
||||
|
||||
impl Doc {
|
||||
/// Plain text of a run of inlines, with markup dropped and link labels kept.
|
||||
/// Used for the title fallback and anywhere a format needs a bare string.
|
||||
pub fn plain_text(inline: &[Inline]) -> String {
|
||||
let mut out = String::new();
|
||||
Self::write_plain(inline, &mut out);
|
||||
out
|
||||
}
|
||||
|
||||
fn write_plain(inline: &[Inline], out: &mut String) {
|
||||
for item in inline {
|
||||
match item {
|
||||
Inline::Text(text) | Inline::Code(text) => out.push_str(text),
|
||||
Inline::Emph(inner) | Inline::Strong(inner) | Inline::Strike(inner) => {
|
||||
Self::write_plain(inner, out)
|
||||
}
|
||||
Inline::Link { label, .. } => Self::write_plain(label, out),
|
||||
Inline::Image { alt, .. } => Self::write_plain(alt, out),
|
||||
Inline::SoftBreak | Inline::HardBreak => out.push(' '),
|
||||
// Raw markup is not text; a format that wants it reads the variant.
|
||||
Inline::Html(_) => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn plain_text_flattens_markup_and_keeps_link_labels() {
|
||||
let inline = vec![
|
||||
Inline::Text("a ".into()),
|
||||
Inline::Strong(vec![Inline::Text("b".into())]),
|
||||
Inline::Text(" ".into()),
|
||||
Inline::Link {
|
||||
href: "/x".into(),
|
||||
title: None,
|
||||
label: vec![Inline::Emph(vec![Inline::Text("c".into())])],
|
||||
},
|
||||
Inline::SoftBreak,
|
||||
Inline::Code("d".into()),
|
||||
Inline::Html("<br>".into()),
|
||||
];
|
||||
assert_eq!(Doc::plain_text(&inline), "a b c d");
|
||||
}
|
||||
}
|
||||
|
|
@ -7,8 +7,11 @@
|
|||
pub mod cache;
|
||||
pub mod config;
|
||||
pub mod error;
|
||||
pub mod ir;
|
||||
pub mod mime;
|
||||
pub mod parse;
|
||||
pub mod path;
|
||||
pub mod preprocess;
|
||||
pub mod site;
|
||||
pub mod siteset;
|
||||
|
||||
|
|
|
|||
655
core/src/parse.rs
Normal file
655
core/src/parse.rs
Normal file
|
|
@ -0,0 +1,655 @@
|
|||
//! Lowering Markdown into [`Doc`].
|
||||
//!
|
||||
//! `pulldown-cmark` is a pull parser whose events borrow from the source, which
|
||||
//! maps directly onto the flat block sequence the formats want. The alternatives
|
||||
//! build an arena AST that would only be flattened again.
|
||||
//!
|
||||
//! Only tables and strikethrough are enabled beyond CommonMark, matching what the
|
||||
//! Python parsers actually handle. Events belonging to extensions that are off
|
||||
//! are therefore never produced.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
use pulldown_cmark::{Alignment, CodeBlockKind, Event, Options, Parser, Tag, TagEnd};
|
||||
|
||||
use crate::error::Error;
|
||||
use crate::ir::{Align, Block, Doc, Inline};
|
||||
use crate::preprocess::{self, Segment};
|
||||
|
||||
/// Read a source file, expand its directives, and parse the result.
|
||||
pub fn document(source: &Path, root: &Path) -> Result<Doc, Error> {
|
||||
Ok(from_segments(&preprocess::expand(source, root)?))
|
||||
}
|
||||
|
||||
/// Assemble a document from preprocessed segments.
|
||||
///
|
||||
/// A card break or an art block becomes a block directly, never passing through
|
||||
/// the Markdown parser. That is what keeps a `{.card}` directive from surfacing
|
||||
/// as literal paragraph text, which is a bug smolweb works around by stripping
|
||||
/// the line before one of its two parsers sees it.
|
||||
pub fn from_segments(segments: &[Segment]) -> Doc {
|
||||
let mut doc = Doc { blocks: Vec::new(), first_h1: None };
|
||||
for segment in segments {
|
||||
match segment {
|
||||
Segment::Markdown(text) => {
|
||||
let mut parsed = markdown(text);
|
||||
if doc.first_h1.is_none() {
|
||||
doc.first_h1 = parsed.first_h1.take();
|
||||
}
|
||||
doc.blocks.append(&mut parsed.blocks);
|
||||
}
|
||||
Segment::Art { block_type, name, align, lines } => doc.blocks.push(Block::Art {
|
||||
block_type: block_type.clone(),
|
||||
name: name.clone(),
|
||||
align: *align,
|
||||
lines: lines.clone(),
|
||||
}),
|
||||
Segment::CardBreak { title } => {
|
||||
doc.blocks.push(Block::CardBreak { title: title.clone() })
|
||||
}
|
||||
}
|
||||
}
|
||||
doc
|
||||
}
|
||||
|
||||
/// Parse one run of Markdown.
|
||||
pub fn markdown(text: &str) -> Doc {
|
||||
let options = Options::ENABLE_TABLES | Options::ENABLE_STRIKETHROUGH;
|
||||
let mut builder = Builder::default();
|
||||
for event in Parser::new_ext(text, options) {
|
||||
builder.handle(event);
|
||||
}
|
||||
builder.finish()
|
||||
}
|
||||
|
||||
/// Data a tag needs when it closes, for the tags whose `TagEnd` carries none.
|
||||
enum Pending {
|
||||
Heading(u8),
|
||||
Link { href: String, title: Option<String> },
|
||||
Image { src: String, title: Option<String> },
|
||||
}
|
||||
|
||||
struct ListFrame {
|
||||
ordered: bool,
|
||||
start: u64,
|
||||
items: Vec<Vec<Block>>,
|
||||
}
|
||||
|
||||
struct TableFrame {
|
||||
alignments: Vec<Option<Align>>,
|
||||
head: Vec<Vec<Inline>>,
|
||||
rows: Vec<Vec<Vec<Inline>>>,
|
||||
/// The row being filled, whether it belongs to the head or the body.
|
||||
row: Vec<Vec<Inline>>,
|
||||
in_head: bool,
|
||||
}
|
||||
|
||||
struct CodeFrame {
|
||||
info: Option<String>,
|
||||
text: String,
|
||||
}
|
||||
|
||||
struct Builder {
|
||||
/// Stack of open block containers; index 0 is the document body.
|
||||
blocks: Vec<Vec<Block>>,
|
||||
/// Stack of open inline containers, for nested emphasis and link labels.
|
||||
inlines: Vec<Vec<Inline>>,
|
||||
pending: Vec<Pending>,
|
||||
lists: Vec<ListFrame>,
|
||||
tables: Vec<TableFrame>,
|
||||
code: Option<CodeFrame>,
|
||||
/// Accumulating the text of an open HTML block.
|
||||
html: Option<String>,
|
||||
/// Whether the innermost inline container was opened implicitly. A tight
|
||||
/// list item holds its text with no `Paragraph` around it, so inline events
|
||||
/// arrive with nothing open; without this they would be dropped.
|
||||
implicit: bool,
|
||||
first_h1: Option<String>,
|
||||
}
|
||||
|
||||
impl Default for Builder {
|
||||
fn default() -> Self {
|
||||
Builder {
|
||||
blocks: vec![Vec::new()],
|
||||
inlines: Vec::new(),
|
||||
pending: Vec::new(),
|
||||
lists: Vec::new(),
|
||||
tables: Vec::new(),
|
||||
code: None,
|
||||
html: None,
|
||||
implicit: false,
|
||||
first_h1: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Builder {
|
||||
fn finish(mut self) -> Doc {
|
||||
self.flush_implicit();
|
||||
// Start and end events are balanced, so only the document frame is left.
|
||||
let blocks = self.blocks.pop().unwrap_or_default();
|
||||
Doc { blocks, first_h1: self.first_h1 }
|
||||
}
|
||||
|
||||
fn handle(&mut self, event: Event<'_>) {
|
||||
match event {
|
||||
Event::Start(tag) => self.start(tag),
|
||||
Event::End(end) => self.end(end),
|
||||
Event::Text(text) => match &mut self.code {
|
||||
Some(frame) => frame.text.push_str(&text),
|
||||
None => self.push_inline(Inline::Text(text.to_string())),
|
||||
},
|
||||
Event::Code(code) => self.push_inline(Inline::Code(code.to_string())),
|
||||
Event::SoftBreak => self.push_inline(Inline::SoftBreak),
|
||||
Event::HardBreak => self.push_inline(Inline::HardBreak),
|
||||
Event::Rule => self.push_block(Block::Rule),
|
||||
Event::Html(html) => match &mut self.html {
|
||||
Some(open) => open.push_str(&html),
|
||||
None => self.push_block(Block::Html(html.to_string())),
|
||||
},
|
||||
Event::InlineHtml(html) => self.push_inline(Inline::Html(html.to_string())),
|
||||
// Math, footnotes and task markers need extensions that are off.
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
fn start(&mut self, tag: Tag<'_>) {
|
||||
// An implicit paragraph ends where the next block begins, so it has to
|
||||
// be closed before that block is added or the order would invert.
|
||||
if matches!(
|
||||
tag,
|
||||
Tag::Paragraph
|
||||
| Tag::BlockQuote(_)
|
||||
| Tag::CodeBlock(_)
|
||||
| Tag::HtmlBlock
|
||||
| Tag::List(_)
|
||||
| Tag::Item
|
||||
| Tag::Table(_)
|
||||
) {
|
||||
self.flush_implicit();
|
||||
}
|
||||
match tag {
|
||||
Tag::Paragraph => self.open_inlines(),
|
||||
Tag::Heading { level, .. } => {
|
||||
self.pending.push(Pending::Heading(level as u8));
|
||||
self.open_inlines();
|
||||
}
|
||||
Tag::BlockQuote(_) => self.blocks.push(Vec::new()),
|
||||
Tag::CodeBlock(kind) => {
|
||||
let info = match kind {
|
||||
CodeBlockKind::Fenced(info) if !info.is_empty() => Some(info.to_string()),
|
||||
_ => None,
|
||||
};
|
||||
self.code = Some(CodeFrame { info, text: String::new() });
|
||||
}
|
||||
Tag::HtmlBlock => self.html = Some(String::new()),
|
||||
Tag::List(start) => self.lists.push(ListFrame {
|
||||
ordered: start.is_some(),
|
||||
start: start.unwrap_or(1),
|
||||
items: Vec::new(),
|
||||
}),
|
||||
Tag::Item => self.blocks.push(Vec::new()),
|
||||
Tag::Table(alignments) => self.tables.push(TableFrame {
|
||||
alignments: alignments.into_iter().map(align_of).collect(),
|
||||
head: Vec::new(),
|
||||
rows: Vec::new(),
|
||||
row: Vec::new(),
|
||||
in_head: false,
|
||||
}),
|
||||
Tag::TableHead => {
|
||||
if let Some(table) = self.tables.last_mut() {
|
||||
table.in_head = true;
|
||||
table.row.clear();
|
||||
}
|
||||
}
|
||||
Tag::TableRow => {
|
||||
if let Some(table) = self.tables.last_mut() {
|
||||
table.row.clear();
|
||||
}
|
||||
}
|
||||
Tag::TableCell => self.open_inlines(),
|
||||
Tag::Emphasis | Tag::Strong | Tag::Strikethrough => self.open_inlines(),
|
||||
Tag::Link { dest_url, title, .. } => {
|
||||
self.pending
|
||||
.push(Pending::Link { href: dest_url.to_string(), title: non_empty(&title) });
|
||||
self.open_inlines();
|
||||
}
|
||||
Tag::Image { dest_url, title, .. } => {
|
||||
self.pending
|
||||
.push(Pending::Image { src: dest_url.to_string(), title: non_empty(&title) });
|
||||
self.open_inlines();
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
fn end(&mut self, end: TagEnd) {
|
||||
match end {
|
||||
TagEnd::Paragraph => {
|
||||
let inline = self.close_inlines();
|
||||
self.push_block(Block::Paragraph(inline));
|
||||
}
|
||||
TagEnd::Heading(_) => {
|
||||
let inline = self.close_inlines();
|
||||
let level = match self.pending.pop() {
|
||||
Some(Pending::Heading(level)) => level,
|
||||
other => {
|
||||
if let Some(value) = other {
|
||||
self.pending.push(value);
|
||||
}
|
||||
1
|
||||
}
|
||||
};
|
||||
if level == 1 && self.first_h1.is_none() {
|
||||
self.first_h1 = Some(Doc::plain_text(&inline));
|
||||
}
|
||||
self.push_block(Block::Heading { level, inline });
|
||||
}
|
||||
TagEnd::BlockQuote(_) => {
|
||||
self.flush_implicit();
|
||||
let blocks = self.blocks.pop().unwrap_or_default();
|
||||
self.push_block(Block::BlockQuote(blocks));
|
||||
}
|
||||
TagEnd::CodeBlock => {
|
||||
if let Some(frame) = self.code.take() {
|
||||
let lines = frame.text.lines().map(str::to_string).collect();
|
||||
self.push_block(Block::CodeBlock { info: frame.info, lines });
|
||||
}
|
||||
}
|
||||
TagEnd::HtmlBlock => {
|
||||
if let Some(html) = self.html.take() {
|
||||
self.push_block(Block::Html(html));
|
||||
}
|
||||
}
|
||||
TagEnd::List(_) => {
|
||||
self.flush_implicit();
|
||||
if let Some(frame) = self.lists.pop() {
|
||||
self.push_block(Block::List {
|
||||
ordered: frame.ordered,
|
||||
start: frame.start,
|
||||
items: frame.items,
|
||||
});
|
||||
}
|
||||
}
|
||||
TagEnd::Item => {
|
||||
self.flush_implicit();
|
||||
let item = self.blocks.pop().unwrap_or_default();
|
||||
if let Some(list) = self.lists.last_mut() {
|
||||
list.items.push(item);
|
||||
}
|
||||
}
|
||||
TagEnd::Table => {
|
||||
if let Some(frame) = self.tables.pop() {
|
||||
self.push_block(Block::Table {
|
||||
alignments: frame.alignments,
|
||||
head: frame.head,
|
||||
rows: frame.rows,
|
||||
});
|
||||
}
|
||||
}
|
||||
TagEnd::TableHead => {
|
||||
if let Some(table) = self.tables.last_mut() {
|
||||
table.head = std::mem::take(&mut table.row);
|
||||
table.in_head = false;
|
||||
}
|
||||
}
|
||||
TagEnd::TableRow => {
|
||||
if let Some(table) = self.tables.last_mut() {
|
||||
let row = std::mem::take(&mut table.row);
|
||||
table.rows.push(row);
|
||||
}
|
||||
}
|
||||
TagEnd::TableCell => {
|
||||
let cell = self.close_inlines();
|
||||
if let Some(table) = self.tables.last_mut() {
|
||||
table.row.push(cell);
|
||||
}
|
||||
}
|
||||
TagEnd::Emphasis => {
|
||||
let inner = self.close_inlines();
|
||||
self.push_inline(Inline::Emph(inner));
|
||||
}
|
||||
TagEnd::Strong => {
|
||||
let inner = self.close_inlines();
|
||||
self.push_inline(Inline::Strong(inner));
|
||||
}
|
||||
TagEnd::Strikethrough => {
|
||||
let inner = self.close_inlines();
|
||||
self.push_inline(Inline::Strike(inner));
|
||||
}
|
||||
TagEnd::Link => {
|
||||
let label = self.close_inlines();
|
||||
if let Some(Pending::Link { href, title }) = self.pending.pop() {
|
||||
self.push_inline(Inline::Link { href, title, label });
|
||||
}
|
||||
}
|
||||
TagEnd::Image => {
|
||||
let alt = self.close_inlines();
|
||||
if let Some(Pending::Image { src, title }) = self.pending.pop() {
|
||||
self.push_inline(Inline::Image { src, title, alt });
|
||||
}
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
}
|
||||
|
||||
fn open_inlines(&mut self) {
|
||||
self.inlines.push(Vec::new());
|
||||
}
|
||||
|
||||
fn close_inlines(&mut self) -> Vec<Inline> {
|
||||
self.inlines.pop().unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Add an inline to the innermost open inline container, opening an implicit
|
||||
/// paragraph if nothing is open. That happens for every tight list item,
|
||||
/// whose text pulldown-cmark emits with no `Paragraph` around it.
|
||||
fn push_inline(&mut self, inline: Inline) {
|
||||
if self.inlines.is_empty() {
|
||||
self.inlines.push(Vec::new());
|
||||
self.implicit = true;
|
||||
}
|
||||
if let Some(frame) = self.inlines.last_mut() {
|
||||
frame.push(inline);
|
||||
}
|
||||
}
|
||||
|
||||
/// Close an implicit paragraph, if one is open, into the current block
|
||||
/// container. Pushes directly rather than through `push_block`, which calls
|
||||
/// this.
|
||||
fn flush_implicit(&mut self) {
|
||||
if !self.implicit {
|
||||
return;
|
||||
}
|
||||
self.implicit = false;
|
||||
let inline = self.inlines.pop().unwrap_or_default();
|
||||
if !inline.is_empty()
|
||||
&& let Some(frame) = self.blocks.last_mut()
|
||||
{
|
||||
frame.push(Block::Paragraph(inline));
|
||||
}
|
||||
}
|
||||
|
||||
fn push_block(&mut self, block: Block) {
|
||||
self.flush_implicit();
|
||||
if let Some(frame) = self.blocks.last_mut() {
|
||||
frame.push(block);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn align_of(alignment: Alignment) -> Option<Align> {
|
||||
match alignment {
|
||||
Alignment::None => None,
|
||||
Alignment::Left => Some(Align::Left),
|
||||
Alignment::Center => Some(Align::Center),
|
||||
Alignment::Right => Some(Align::Right),
|
||||
}
|
||||
}
|
||||
|
||||
fn non_empty(value: &str) -> Option<String> {
|
||||
(!value.is_empty()).then(|| value.to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn blocks(text: &str) -> Vec<Block> {
|
||||
markdown(text).blocks
|
||||
}
|
||||
|
||||
fn text(value: &str) -> Vec<Inline> {
|
||||
vec![Inline::Text(value.into())]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_headings_and_paragraphs() {
|
||||
assert_eq!(
|
||||
blocks("# Title\n\nBody.\n"),
|
||||
vec![
|
||||
Block::Heading { level: 1, inline: text("Title") },
|
||||
Block::Paragraph(text("Body.")),
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_setext_headings() {
|
||||
// wapdown's parser handles these and md2txt's does not, so unifying the
|
||||
// two parsers gains them for the text formats.
|
||||
assert_eq!(
|
||||
blocks("Title\n=====\n"),
|
||||
vec![Block::Heading { level: 1, inline: text("Title") }]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn records_the_first_level_one_heading() {
|
||||
let doc = markdown("## Second\n\n# First\n\n# Another\n");
|
||||
assert_eq!(doc.first_h1.as_deref(), Some("First"));
|
||||
// A document with no h1 has nothing to derive a title from.
|
||||
assert_eq!(markdown("## Only\n").first_h1, None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_inline_markup() {
|
||||
let Block::Paragraph(inline) = &blocks("a *b* **c** ~~d~~ `e`\n")[0] else {
|
||||
panic!("expected a paragraph")
|
||||
};
|
||||
assert_eq!(
|
||||
inline,
|
||||
&vec![
|
||||
Inline::Text("a ".into()),
|
||||
Inline::Emph(text("b")),
|
||||
Inline::Text(" ".into()),
|
||||
Inline::Strong(text("c")),
|
||||
Inline::Text(" ".into()),
|
||||
Inline::Strike(text("d")),
|
||||
Inline::Text(" ".into()),
|
||||
Inline::Code("e".into()),
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_links_and_images_with_their_labels() {
|
||||
let Block::Paragraph(inline) = &blocks("[a *b*](/x \"T\") \n")[0] else {
|
||||
panic!("expected a paragraph")
|
||||
};
|
||||
assert_eq!(
|
||||
inline[0],
|
||||
Inline::Link {
|
||||
href: "/x".into(),
|
||||
title: Some("T".into()),
|
||||
label: vec![Inline::Text("a ".into()), Inline::Emph(text("b"))],
|
||||
}
|
||||
);
|
||||
assert_eq!(
|
||||
inline[2],
|
||||
Inline::Image { src: "/i.png".into(), title: None, alt: text("alt") }
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_nested_lists() {
|
||||
let parsed = blocks("- a\n - b\n- c\n");
|
||||
let Block::List { ordered, start, items } = &parsed[0] else { panic!("expected a list") };
|
||||
assert!(!ordered);
|
||||
assert_eq!(*start, 1);
|
||||
assert_eq!(items.len(), 2);
|
||||
// The first item holds its own paragraph and the nested list.
|
||||
assert_eq!(items[0][0], Block::Paragraph(text("a")));
|
||||
assert!(matches!(items[0][1], Block::List { .. }));
|
||||
assert_eq!(items[1][0], Block::Paragraph(text("c")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_tight_list_items_text_is_kept() {
|
||||
// Regression: pulldown-cmark emits a tight item's text with no Paragraph
|
||||
// around it, so an implicit one has to be opened or the text is dropped.
|
||||
let Block::List { items, .. } = &blocks("- a\n- b\n")[0] else { panic!("expected a list") };
|
||||
assert_eq!(
|
||||
items,
|
||||
&vec![vec![Block::Paragraph(text("a"))], vec![Block::Paragraph(text("b"))]]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_tight_item_keeps_its_inline_markup_in_order() {
|
||||
let Block::List { items, .. } = &blocks("- a *b* c\n")[0] else {
|
||||
panic!("expected a list")
|
||||
};
|
||||
assert_eq!(
|
||||
items[0],
|
||||
vec![Block::Paragraph(vec![
|
||||
Inline::Text("a ".into()),
|
||||
Inline::Emph(text("b")),
|
||||
Inline::Text(" c".into()),
|
||||
])]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_tight_item_opening_with_markup_is_still_one_paragraph() {
|
||||
let Block::List { items, .. } = &blocks("- *a* b\n")[0] else { panic!("expected a list") };
|
||||
assert_eq!(
|
||||
items[0],
|
||||
vec![Block::Paragraph(vec![Inline::Emph(text("a")), Inline::Text(" b".into()),])]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_loose_list_is_shaped_the_same_as_a_tight_one() {
|
||||
// The distinction is pulldown-cmark's, not ours: both yield a paragraph
|
||||
// per item, so no format has to know which it was.
|
||||
let Block::List { items: tight, .. } = &blocks("- a\n- b\n")[0] else { panic!("list") };
|
||||
let Block::List { items: loose, .. } = &blocks("- a\n\n- b\n")[0] else { panic!("list") };
|
||||
assert_eq!(tight, loose);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_ordered_list_keeps_its_starting_number() {
|
||||
let Block::List { ordered, start, items } = &blocks("3. a\n4. b\n")[0] else {
|
||||
panic!("expected a list")
|
||||
};
|
||||
assert!(ordered);
|
||||
assert_eq!(*start, 3);
|
||||
assert_eq!(items.len(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_nested_blockquotes() {
|
||||
// Depth comes from the nesting, so there is one place it is recorded.
|
||||
let parsed = blocks("> outer\n>\n> > inner\n");
|
||||
let Block::BlockQuote(outer) = &parsed[0] else { panic!("expected a quote") };
|
||||
assert_eq!(outer[0], Block::Paragraph(text("outer")));
|
||||
assert_eq!(outer[1], Block::BlockQuote(vec![Block::Paragraph(text("inner"))]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_fenced_and_indented_code() {
|
||||
assert_eq!(
|
||||
blocks("```rust\nlet x = 1;\nlet y = 2;\n```\n"),
|
||||
vec![Block::CodeBlock {
|
||||
info: Some("rust".into()),
|
||||
lines: vec!["let x = 1;".into(), "let y = 2;".into()],
|
||||
}]
|
||||
);
|
||||
assert_eq!(
|
||||
blocks(" indented\n"),
|
||||
vec![Block::CodeBlock { info: None, lines: vec!["indented".into()] }]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_tables_with_their_alignments() {
|
||||
let parsed = blocks("| a | b |\n| :-- | --: |\n| 1 | 2 |\n| 3 | 4 |\n");
|
||||
let Block::Table { alignments, head, rows } = &parsed[0] else {
|
||||
panic!("expected a table")
|
||||
};
|
||||
assert_eq!(alignments, &vec![Some(Align::Left), Some(Align::Right)]);
|
||||
assert_eq!(head, &vec![text("a"), text("b")]);
|
||||
assert_eq!(rows, &vec![vec![text("1"), text("2")], vec![text("3"), text("4")]]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_rules_and_breaks() {
|
||||
assert_eq!(blocks("---\n"), vec![Block::Rule]);
|
||||
let Block::Paragraph(inline) = &blocks("a\nb\n")[0] else { panic!("expected a paragraph") };
|
||||
assert_eq!(inline[1], Inline::SoftBreak);
|
||||
let Block::Paragraph(inline) = &blocks("a \nb\n")[0] else {
|
||||
panic!("expected a paragraph")
|
||||
};
|
||||
assert_eq!(inline[1], Inline::HardBreak);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn keeps_raw_html_rather_than_dropping_it() {
|
||||
// XHTML-MP passes it through and the text formats strip it; the parser
|
||||
// does not get to decide.
|
||||
let parsed = blocks("<div>x</div>\n");
|
||||
assert!(matches!(&parsed[0], Block::Html(html) if html.contains("<div>")));
|
||||
let Block::Paragraph(inline) = &blocks("a <b>c</b>\n")[0] else {
|
||||
panic!("expected a paragraph")
|
||||
};
|
||||
assert_eq!(inline[1], Inline::Html("<b>".into()));
|
||||
}
|
||||
|
||||
// -- Segment assembly --------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn card_breaks_and_art_become_blocks_without_passing_through_the_parser() {
|
||||
let segments = vec![
|
||||
Segment::Markdown("Intro.\n".into()),
|
||||
Segment::CardBreak { title: Some("Weather".into()) },
|
||||
Segment::Markdown("Cold.\n".into()),
|
||||
Segment::Art {
|
||||
block_type: "banner".into(),
|
||||
name: "Logo".into(),
|
||||
align: Some(Align::Center),
|
||||
lines: vec!["/\\".into()],
|
||||
},
|
||||
];
|
||||
let doc = from_segments(&segments);
|
||||
assert_eq!(
|
||||
doc.blocks,
|
||||
vec![
|
||||
Block::Paragraph(text("Intro.")),
|
||||
Block::CardBreak { title: Some("Weather".into()) },
|
||||
Block::Paragraph(text("Cold.")),
|
||||
Block::Art {
|
||||
block_type: "banner".into(),
|
||||
name: "Logo".into(),
|
||||
align: Some(Align::Center),
|
||||
lines: vec!["/\\".into()],
|
||||
},
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_card_directive_never_appears_as_text() {
|
||||
// The leak smolweb works around by stripping the line before md2txt sees
|
||||
// it: here the directive cannot reach the Markdown parser at all.
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let root = dir.path().canonicalize().unwrap();
|
||||
std::fs::write(root.join("page.md"), "Intro.\n\n{.card Weather}\nCold.\n").unwrap();
|
||||
|
||||
let doc = document(&root.join("page.md"), &root).unwrap();
|
||||
let rendered = format!("{:?}", doc.blocks);
|
||||
assert!(!rendered.contains("{.card"), "{rendered}");
|
||||
assert!(doc.blocks.contains(&Block::CardBreak { title: Some("Weather".into()) }));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_title_comes_from_the_first_markdown_run_that_has_one() {
|
||||
let segments = vec![
|
||||
Segment::CardBreak { title: None },
|
||||
Segment::Markdown("## not it\n".into()),
|
||||
Segment::Markdown("# Found\n".into()),
|
||||
Segment::Markdown("# Later\n".into()),
|
||||
];
|
||||
assert_eq!(from_segments(&segments).first_h1.as_deref(), Some("Found"));
|
||||
}
|
||||
}
|
||||
537
core/src/preprocess.rs
Normal file
537
core/src/preprocess.rs
Normal file
|
|
@ -0,0 +1,537 @@
|
|||
//! The line-level pass that runs before Markdown parsing.
|
||||
//!
|
||||
//! Four directives live outside CommonMark, and all four occupy a whole line, so
|
||||
//! they are recognised here rather than by extending the Markdown parser. That
|
||||
//! is also how the Python does it; the difference is that this pass emits typed
|
||||
//! segments instead of encoding them as sentinel strings inside the line stream.
|
||||
//!
|
||||
//! This module is the only part of the rendering pipeline that opens files.
|
||||
//! Keeping it so means the root-containment check has exactly one home, which is
|
||||
//! what closes the traversal smolweb has: md2txt resolves an include target and
|
||||
//! checks only that it exists, so `{.include ../../../../etc/passwd}` in any
|
||||
//! served document reads and emits that file.
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
use std::fs;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
use crate::error::{Error, IncludeReason};
|
||||
use crate::ir::Align;
|
||||
|
||||
/// How deep includes may nest. The cycle set alone does not bound a long chain
|
||||
/// that never repeats a file.
|
||||
const MAX_DEPTH: usize = 16;
|
||||
/// Caps on the expanded result. The cycle set is per-*stack*, so a diamond —
|
||||
/// `a` includes `b` and `c`, both include `d` — fans out exponentially without
|
||||
/// ever repeating a file on one path. smolweb has nothing that stops this.
|
||||
const MAX_LINES: usize = 200_000;
|
||||
const MAX_BYTES: usize = 8 * 1024 * 1024;
|
||||
|
||||
/// A run of input, classified. Markdown runs are parsed; the rest become blocks
|
||||
/// directly, so no directive can ever be mistaken for paragraph text.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
pub enum Segment {
|
||||
Markdown(String),
|
||||
Art { block_type: String, name: String, align: Option<Align>, lines: Vec<String> },
|
||||
CardBreak { title: Option<String> },
|
||||
}
|
||||
|
||||
/// Read `source` and expand it into segments.
|
||||
///
|
||||
/// `root` bounds every path this may read: an include or art target that
|
||||
/// canonicalises outside it is refused, not followed.
|
||||
pub fn expand(source: &Path, root: &Path) -> Result<Vec<Segment>, Error> {
|
||||
let mut out = Segments::default();
|
||||
let mut stack = BTreeSet::new();
|
||||
let canonical = canonical_within(source, root)?;
|
||||
stack.insert(canonical.clone());
|
||||
expand_file(&canonical, root, &mut stack, 0, &mut out)?;
|
||||
Ok(out.finish())
|
||||
}
|
||||
|
||||
/// Accumulates segments, merging consecutive Markdown lines into one run so the
|
||||
/// Markdown parser sees whole constructs rather than line fragments.
|
||||
#[derive(Default)]
|
||||
struct Segments {
|
||||
done: Vec<Segment>,
|
||||
markdown: String,
|
||||
lines: usize,
|
||||
bytes: usize,
|
||||
}
|
||||
|
||||
impl Segments {
|
||||
fn push_line(&mut self, line: &str) -> Result<(), Error> {
|
||||
self.lines += 1;
|
||||
self.bytes += line.len();
|
||||
if self.lines > MAX_LINES || self.bytes > MAX_BYTES {
|
||||
return Err(Error::Include { path: PathBuf::new(), reason: IncludeReason::TooLarge });
|
||||
}
|
||||
self.markdown.push_str(line);
|
||||
self.markdown.push('\n');
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn push(&mut self, segment: Segment) {
|
||||
self.flush();
|
||||
self.done.push(segment);
|
||||
}
|
||||
|
||||
fn flush(&mut self) {
|
||||
if !self.markdown.is_empty() {
|
||||
self.done.push(Segment::Markdown(std::mem::take(&mut self.markdown)));
|
||||
}
|
||||
}
|
||||
|
||||
fn finish(mut self) -> Vec<Segment> {
|
||||
self.flush();
|
||||
self.done
|
||||
}
|
||||
}
|
||||
|
||||
fn expand_file(
|
||||
path: &Path,
|
||||
root: &Path,
|
||||
stack: &mut BTreeSet<PathBuf>,
|
||||
depth: usize,
|
||||
out: &mut Segments,
|
||||
) -> Result<(), Error> {
|
||||
let text =
|
||||
fs::read_to_string(path).map_err(|cause| Error::Io { path: path.to_path_buf(), cause })?;
|
||||
let base = path.parent().unwrap_or(root);
|
||||
|
||||
for line in text.lines() {
|
||||
if let Some(title) = card_break(line) {
|
||||
out.push(Segment::CardBreak { title: title.map(str::to_string) });
|
||||
continue;
|
||||
}
|
||||
if let Some(pieces) = art_pieces(line) {
|
||||
for (label, target) in pieces {
|
||||
let (block_type, name, align) = parse_label(label);
|
||||
let file = resolve(target, base, root)?;
|
||||
let art = fs::read_to_string(&file)
|
||||
.map_err(|cause| Error::Io { path: file.clone(), cause })?;
|
||||
out.push(Segment::Art {
|
||||
block_type,
|
||||
name,
|
||||
align,
|
||||
lines: art.lines().map(str::to_string).collect(),
|
||||
});
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if let Some(target) = include_target(line) {
|
||||
if depth + 1 > MAX_DEPTH {
|
||||
return Err(Error::Include {
|
||||
path: path.to_path_buf(),
|
||||
reason: IncludeReason::TooDeep,
|
||||
});
|
||||
}
|
||||
let file = resolve(target, base, root)?;
|
||||
if !stack.insert(file.clone()) {
|
||||
return Err(Error::Include { path: file, reason: IncludeReason::Circular });
|
||||
}
|
||||
expand_file(&file, root, stack, depth + 1, out)?;
|
||||
stack.remove(&file);
|
||||
continue;
|
||||
}
|
||||
out.push_line(line)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Resolve a directive's target relative to the including file, and require the
|
||||
/// result to be a readable file inside the content root.
|
||||
fn resolve(target: &str, base: &Path, root: &Path) -> Result<PathBuf, Error> {
|
||||
canonical_within(&base.join(target), root)
|
||||
}
|
||||
|
||||
fn canonical_within(path: &Path, root: &Path) -> Result<PathBuf, Error> {
|
||||
let canonical = path
|
||||
.canonicalize()
|
||||
.map_err(|_| Error::Include { path: path.to_path_buf(), reason: IncludeReason::Missing })?;
|
||||
if !canonical.starts_with(root) || !canonical.is_file() {
|
||||
return Err(Error::Include { path: canonical, reason: IncludeReason::Outside });
|
||||
}
|
||||
Ok(canonical)
|
||||
}
|
||||
|
||||
// -- Directive recognition ------------------------------------------------
|
||||
//
|
||||
// Hand-written rather than regex-driven: each pattern is an anchored prefix and
|
||||
// suffix around one capture, and a dependency for five of those is not a trade
|
||||
// worth making in a codebase meant to stay auditable.
|
||||
|
||||
/// `{.card}` or `{.card Title}` alone on a line. The outer `Option` is whether
|
||||
/// the line is a card break at all; the inner one is whether it carries a title.
|
||||
pub fn card_break(line: &str) -> Option<Option<&str>> {
|
||||
let rest = line.trim().strip_prefix('{')?.trim_start().strip_prefix(".card")?;
|
||||
let inner = rest.trim_end().strip_suffix('}')?;
|
||||
if inner.is_empty() {
|
||||
return Some(None);
|
||||
}
|
||||
// A following character must be whitespace, or this is `{.cardsomething}`.
|
||||
if !inner.starts_with(char::is_whitespace) {
|
||||
return None;
|
||||
}
|
||||
let title = inner.trim();
|
||||
Some((!title.is_empty()).then_some(title))
|
||||
}
|
||||
|
||||
/// `![[target]]` or `{.include target}` alone on a line.
|
||||
pub fn include_target(line: &str) -> Option<&str> {
|
||||
let trimmed = line.trim();
|
||||
if let Some(inner) = trimmed.strip_prefix("![[").and_then(|r| r.strip_suffix("]]")) {
|
||||
return unquote(inner);
|
||||
}
|
||||
let rest = trimmed.strip_prefix('{')?.trim_start().strip_prefix(".include")?;
|
||||
if !rest.starts_with(char::is_whitespace) {
|
||||
return None;
|
||||
}
|
||||
unquote(rest.trim_end().strip_suffix('}')?)
|
||||
}
|
||||
|
||||
/// Strip surrounding matched quotes, as the Python's target normalisation does.
|
||||
fn unquote(value: &str) -> Option<&str> {
|
||||
let trimmed = value.trim();
|
||||
let inner = match (trimmed.chars().next(), trimmed.chars().last()) {
|
||||
(Some(first @ ('\'' | '"')), Some(last)) if first == last && trimmed.len() >= 2 => {
|
||||
trimmed[1..trimmed.len() - 1].trim()
|
||||
}
|
||||
_ => trimmed,
|
||||
};
|
||||
(!inner.is_empty()).then_some(inner)
|
||||
}
|
||||
|
||||
/// `#[label](target)` pieces filling a whole line, as `(label, target)` pairs.
|
||||
///
|
||||
/// One piece is the block form, several the inline form; the Python accepts the
|
||||
/// inline form only when nothing but whitespace surrounds the pieces, which is
|
||||
/// what makes this a block-level directive and keeps it from splitting a
|
||||
/// sentence.
|
||||
pub fn art_pieces(line: &str) -> Option<Vec<(&str, &str)>> {
|
||||
let mut rest = line.trim();
|
||||
// The block form allows a trailing MultiMarkdown attribute list. It is
|
||||
// recognised so the line is still treated as art, but no format reads it.
|
||||
if let Some(open) = rest.rfind("{:")
|
||||
&& rest.ends_with('}')
|
||||
{
|
||||
rest = rest[..open].trim_end();
|
||||
}
|
||||
|
||||
let mut pieces = Vec::new();
|
||||
while !rest.is_empty() {
|
||||
let (label, target, consumed) = one_art(rest)?;
|
||||
pieces.push((label, target));
|
||||
rest = rest[consumed..].trim_start();
|
||||
}
|
||||
(!pieces.is_empty()).then_some(pieces)
|
||||
}
|
||||
|
||||
/// One `#[label](target)` at the start of `text`, with the length it consumed.
|
||||
fn one_art(text: &str) -> Option<(&str, &str, usize)> {
|
||||
let after_open = text.strip_prefix("#[")?;
|
||||
let label_end = after_open.find(']')?;
|
||||
let label = &after_open[..label_end];
|
||||
let after_label = &after_open[label_end + 1..];
|
||||
let after_paren = after_label.strip_prefix('(')?;
|
||||
let target_end = after_paren.find(')')?;
|
||||
let target = &after_paren[..target_end];
|
||||
if label.is_empty() || target.is_empty() {
|
||||
return None;
|
||||
}
|
||||
// "#[" + label + "](" + target + ")"
|
||||
let consumed = 2 + label.len() + 2 + target.len() + 1;
|
||||
Some((label, target, consumed))
|
||||
}
|
||||
|
||||
/// Split an art label into its block type, name and alignment.
|
||||
///
|
||||
/// Tokens beginning with a colon set the alignment; the first remaining token is
|
||||
/// the block type and the rest its name. An absent type is `custom`, matching
|
||||
/// the Python.
|
||||
fn parse_label(label: &str) -> (String, String, Option<Align>) {
|
||||
let mut align = None;
|
||||
let mut words = Vec::new();
|
||||
for token in label.split_whitespace() {
|
||||
match token.strip_prefix(':') {
|
||||
Some(tag) => {
|
||||
align = match tag.to_ascii_lowercase().as_str() {
|
||||
"left" => Some(Align::Left),
|
||||
"right" => Some(Align::Right),
|
||||
"center" | "centre" => Some(Align::Center),
|
||||
// An unrecognised tag is ignored rather than treated as a name.
|
||||
_ => align,
|
||||
};
|
||||
}
|
||||
None => words.push(token),
|
||||
}
|
||||
}
|
||||
let block_type = words.first().copied().unwrap_or("custom").to_string();
|
||||
let name = words.get(1..).map(|rest| rest.join(" ")).unwrap_or_default();
|
||||
(block_type, name, align)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
// -- Directive recognition ---------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn recognises_card_breaks() {
|
||||
assert_eq!(card_break("{.card}"), Some(None));
|
||||
assert_eq!(card_break(" {.card} "), Some(None));
|
||||
assert_eq!(card_break("{ .card }"), Some(None));
|
||||
assert_eq!(card_break("{.card Weather}"), Some(Some("Weather")));
|
||||
assert_eq!(card_break("{ .card Two Words }"), Some(Some("Two Words")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_near_misses_for_card_breaks() {
|
||||
assert_eq!(card_break("{.cardinal}"), None);
|
||||
assert_eq!(card_break("text {.card}"), None);
|
||||
assert_eq!(card_break("{.card"), None);
|
||||
assert_eq!(card_break("{.cards Weather}"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recognises_includes_in_both_spellings() {
|
||||
assert_eq!(include_target("![[notes.md]]"), Some("notes.md"));
|
||||
assert_eq!(include_target(" ![[a/b.md]] "), Some("a/b.md"));
|
||||
assert_eq!(include_target("{.include notes.md}"), Some("notes.md"));
|
||||
assert_eq!(include_target("{ .include notes.md }"), Some("notes.md"));
|
||||
// Quoted targets, as the Python's normalisation allows.
|
||||
assert_eq!(include_target("{.include \"a b.md\"}"), Some("a b.md"));
|
||||
assert_eq!(include_target("![['q.md']]"), Some("q.md"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_near_misses_for_includes() {
|
||||
assert_eq!(include_target("see ![[notes.md]] there"), None);
|
||||
assert_eq!(include_target("{.includes notes.md}"), None);
|
||||
assert_eq!(include_target("{.include}"), None);
|
||||
assert_eq!(include_target("![[]]"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recognises_art_in_block_and_inline_forms() {
|
||||
assert_eq!(art_pieces("#[banner](logo.txt)"), Some(vec![("banner", "logo.txt")]));
|
||||
assert_eq!(
|
||||
art_pieces("#[a](one.txt) #[b](two.txt)"),
|
||||
Some(vec![("a", "one.txt"), ("b", "two.txt")])
|
||||
);
|
||||
// A trailing attribute list keeps the line recognised as art.
|
||||
assert_eq!(art_pieces("#[banner](logo.txt){: .wide}"), Some(vec![("banner", "logo.txt")]));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn art_must_fill_the_whole_line() {
|
||||
// Otherwise it would split a sentence, which is why the Python requires
|
||||
// the surrounding text to be whitespace only.
|
||||
assert_eq!(art_pieces("see #[a](one.txt)"), None);
|
||||
assert_eq!(art_pieces("#[a](one.txt) trailing"), None);
|
||||
assert_eq!(art_pieces("plain text"), None);
|
||||
assert_eq!(art_pieces("#[a]()"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parses_art_labels() {
|
||||
assert_eq!(parse_label("banner"), ("banner".into(), String::new(), None));
|
||||
assert_eq!(parse_label("figlet Big Title"), ("figlet".into(), "Big Title".into(), None));
|
||||
assert_eq!(
|
||||
parse_label(":center banner Logo"),
|
||||
("banner".into(), "Logo".into(), Some(Align::Center))
|
||||
);
|
||||
// The British spelling is accepted and folded onto the same value.
|
||||
assert_eq!(parse_label(":centre x").2, Some(Align::Center));
|
||||
assert_eq!(parse_label(":LEFT x").2, Some(Align::Left));
|
||||
// No type at all falls back to `custom`, as the Python does.
|
||||
assert_eq!(parse_label(":right"), ("custom".into(), String::new(), Some(Align::Right)));
|
||||
// An unknown colon tag is dropped, not taken as the type.
|
||||
assert_eq!(parse_label(":nonsense banner"), ("banner".into(), String::new(), None));
|
||||
}
|
||||
|
||||
// -- Expansion ---------------------------------------------------------
|
||||
|
||||
struct Tree(tempfile::TempDir);
|
||||
|
||||
impl Tree {
|
||||
fn new() -> Self {
|
||||
Tree(tempfile::tempdir().unwrap())
|
||||
}
|
||||
|
||||
fn root(&self) -> PathBuf {
|
||||
self.0.path().canonicalize().unwrap()
|
||||
}
|
||||
|
||||
fn write(&self, rel: &str, body: &str) -> PathBuf {
|
||||
let path = self.0.path().join(rel);
|
||||
if let Some(parent) = path.parent() {
|
||||
fs::create_dir_all(parent).unwrap();
|
||||
}
|
||||
fs::write(&path, body).unwrap();
|
||||
path
|
||||
}
|
||||
|
||||
fn expand(&self, rel: &str) -> Result<Vec<Segment>, Error> {
|
||||
expand(&self.0.path().join(rel), &self.root())
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_file_with_no_directives_is_one_markdown_run() {
|
||||
let tree = Tree::new();
|
||||
tree.write("page.md", "# Title\n\nBody.\n");
|
||||
assert_eq!(
|
||||
tree.expand("page.md").unwrap(),
|
||||
vec![Segment::Markdown("# Title\n\nBody.\n".into())]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn card_breaks_split_the_markdown_runs() {
|
||||
let tree = Tree::new();
|
||||
tree.write("page.md", "Intro.\n\n{.card Weather}\nCold.\n");
|
||||
assert_eq!(
|
||||
tree.expand("page.md").unwrap(),
|
||||
vec![
|
||||
Segment::Markdown("Intro.\n\n".into()),
|
||||
Segment::CardBreak { title: Some("Weather".into()) },
|
||||
Segment::Markdown("Cold.\n".into()),
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_include_splices_the_targets_lines() {
|
||||
let tree = Tree::new();
|
||||
tree.write("part.md", "Shared text.\n");
|
||||
tree.write("page.md", "Before.\n\n{.include part.md}\n\nAfter.\n");
|
||||
// The spliced lines join the surrounding run: an include is a text
|
||||
// substitution, not a structural boundary.
|
||||
assert_eq!(
|
||||
tree.expand("page.md").unwrap(),
|
||||
vec![Segment::Markdown("Before.\n\nShared text.\n\nAfter.\n".into())]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_wikilink_include_resolves_relative_to_the_including_file() {
|
||||
let tree = Tree::new();
|
||||
tree.write("sub/part.md", "Nested shared.\n");
|
||||
tree.write("sub/page.md", "![[part.md]]\n");
|
||||
assert_eq!(
|
||||
tree.expand("sub/page.md").unwrap(),
|
||||
vec![Segment::Markdown("Nested shared.\n".into())]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn art_is_read_into_the_segment() {
|
||||
let tree = Tree::new();
|
||||
tree.write("logo.txt", " /\\ \n/__\\\n");
|
||||
tree.write("page.md", "#[:center banner Logo](logo.txt)\n");
|
||||
assert_eq!(
|
||||
tree.expand("page.md").unwrap(),
|
||||
vec![Segment::Art {
|
||||
block_type: "banner".into(),
|
||||
name: "Logo".into(),
|
||||
align: Some(Align::Center),
|
||||
lines: vec![" /\\ ".into(), "/__\\".into()],
|
||||
}]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_include_outside_the_root_is_refused() {
|
||||
// The traversal smolweb has: md2txt resolves the target and checks only
|
||||
// that it exists, with no containment check at all.
|
||||
let tree = Tree::new();
|
||||
let outside = tree.0.path().parent().unwrap().join("itsybitsy-escape.md");
|
||||
fs::write(&outside, "secret\n").unwrap();
|
||||
tree.write("page.md", "{.include ../itsybitsy-escape.md}\n");
|
||||
|
||||
let err = tree.expand("page.md").unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::Outside, .. }), "{err}");
|
||||
fs::remove_file(outside).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_art_target_outside_the_root_is_refused() {
|
||||
let tree = Tree::new();
|
||||
let outside = tree.0.path().parent().unwrap().join("itsybitsy-art-escape.txt");
|
||||
fs::write(&outside, "secret\n").unwrap();
|
||||
tree.write("page.md", "#[banner x](../itsybitsy-art-escape.txt)\n");
|
||||
|
||||
let err = tree.expand("page.md").unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::Outside, .. }), "{err}");
|
||||
fs::remove_file(outside).unwrap();
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_missing_target_is_reported_not_ignored() {
|
||||
let tree = Tree::new();
|
||||
tree.write("page.md", "{.include absent.md}\n");
|
||||
let err = tree.expand("page.md").unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::Missing, .. }), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_circular_include_is_refused() {
|
||||
let tree = Tree::new();
|
||||
tree.write("a.md", "A\n{.include b.md}\n");
|
||||
tree.write("b.md", "B\n{.include a.md}\n");
|
||||
let err = tree.expand("a.md").unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::Circular, .. }), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_self_include_is_refused() {
|
||||
let tree = Tree::new();
|
||||
tree.write("a.md", "{.include a.md}\n");
|
||||
let err = tree.expand("a.md").unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::Circular, .. }), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_chain_deeper_than_the_cap_is_refused() {
|
||||
// No file repeats, so the cycle set never fires: this is what the depth
|
||||
// cap is for.
|
||||
let tree = Tree::new();
|
||||
for i in 0..=MAX_DEPTH + 1 {
|
||||
tree.write(&format!("n{i}.md"), &format!("line {i}\n{{.include n{}.md}}\n", i + 1));
|
||||
}
|
||||
let err = tree.expand("n0.md").unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::TooDeep, .. }), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_diamond_fan_out_is_stopped_by_the_size_cap() {
|
||||
// Each level doubles and no file repeats on any single path, so neither
|
||||
// the cycle set nor the depth cap catches it. smolweb expands this until
|
||||
// it runs out of memory.
|
||||
let tree = Tree::new();
|
||||
tree.write("leaf.md", &"filler line\n".repeat(64));
|
||||
let mut previous = "leaf.md".to_string();
|
||||
for level in 0..MAX_DEPTH - 2 {
|
||||
let name = format!("l{level}.md");
|
||||
tree.write(&name, &format!("{{.include {previous}}}\n{{.include {previous}}}\n"));
|
||||
previous = name;
|
||||
}
|
||||
let err = tree.expand(&previous).unwrap_err();
|
||||
assert!(matches!(err, Error::Include { reason: IncludeReason::TooLarge, .. }), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn a_repeated_include_on_separate_paths_is_allowed() {
|
||||
// The cycle set is per-stack, so including one shared file twice in
|
||||
// sequence is fine and must not be mistaken for a cycle.
|
||||
let tree = Tree::new();
|
||||
tree.write("part.md", "shared\n");
|
||||
tree.write("page.md", "{.include part.md}\n{.include part.md}\n");
|
||||
assert_eq!(
|
||||
tree.expand("page.md").unwrap(),
|
||||
vec![Segment::Markdown("shared\nshared\n".into())]
|
||||
);
|
||||
}
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue