refactor: express custom directives with standard markdown constructs

This commit is contained in:
randogoth 2026-10-04 19:40:59 +03:00
parent e5583bbd89
commit 46f12b10da
7 changed files with 558 additions and 454 deletions

440
core/src/directives.rs Normal file
View file

@ -0,0 +1,440 @@
//! Interpreting what itsybitsy adds to CommonMark, after parsing.
//!
//! Every construct here is already standard Markdown, so a document carrying it
//! stays valid and readable in any other tool:
//!
//! | Written | Means |
//! | --- | --- |
//! | `<!-- card Title -->` | a card divider; invisible to any other renderer |
//! | `<!-- center -->` | alignment for the block that follows |
//! | `![alt](art.txt)` | an image whose target is text, inlined verbatim |
//!
//! An HTML comment that is not a recognised directive stays a comment. Comments
//! have a legitimate non-directive use, so claiming the whole namespace would
//! silently swallow them; the cost is that a misspelled directive does nothing
//! rather than complaining.
use std::path::{Path, PathBuf};
use std::{fs, mem};
use crate::error::Error;
use crate::ir::{Align, Block, Doc, Inline};
use crate::mime;
use crate::preprocess::canonical_within;
/// Interpret the directives in a parsed document.
///
/// `base` is the directory relative image targets resolve against, and `root`
/// bounds what may be read. Note that includes are already spliced by this
/// point, so an art target written inside an included file resolves relative to
/// the including document rather than to the file it was written in; write such
/// targets root-absolute (`/art/x.txt`) to be unambiguous.
pub fn apply(doc: &mut Doc, base: &Path, root: &Path) -> Result<(), Error> {
doc.blocks = rewrite(mem::take(&mut doc.blocks), base, root)?;
Ok(())
}
fn rewrite(blocks: Vec<Block>, base: &Path, root: &Path) -> Result<Vec<Block>, Error> {
let mut out = Vec::with_capacity(blocks.len());
let mut pending: Option<Directive> = None;
for block in blocks {
if let Block::Html(html) = &block {
match directive(html) {
Some(Directive::Card { title }) => {
out.push(Block::CardBreak { title });
continue;
}
Some(align @ Directive::Align { .. }) => {
pending = Some(align);
continue;
}
None => {}
}
}
let produced = match block {
// A paragraph of nothing but art references becomes those blocks.
Block::Paragraph(inline) => match art(&inline, base, root)? {
Some(blocks) => blocks,
None => vec![Block::Paragraph(inline)],
},
Block::BlockQuote(inner) => vec![Block::BlockQuote(rewrite(inner, base, root)?)],
Block::List { ordered, start, items } => {
let items = items
.into_iter()
.map(|item| rewrite(item, base, root))
.collect::<Result<Vec<_>, _>>()?;
vec![Block::List { ordered, start, items }]
}
other => vec![other],
};
match pending.take() {
Some(Directive::Align { align, margin }) => out.extend(produced.into_iter().map(|b| {
// A block that already carries its own alignment keeps it: the
// more specific marker wins over the one that precedes it.
match b {
aligned @ Block::Aligned { .. } => aligned,
other => Block::Aligned { align, margin, block: Box::new(other) },
}
})),
_ => out.extend(produced),
}
}
Ok(out)
}
enum Directive {
Card { title: Option<String> },
Align { align: Align, margin: Option<u16> },
}
/// Recognise a directive in the text of an HTML block, or `None` for an ordinary
/// comment or any other HTML.
fn directive(html: &str) -> Option<Directive> {
let inner = html.trim().strip_prefix("<!--")?.strip_suffix("-->")?;
let mut words = inner.split_whitespace();
let name = words.next()?;
if name == "card" {
let title = words.collect::<Vec<_>>().join(" ");
return Some(Directive::Card { title: (!title.is_empty()).then_some(title) });
}
let align = align_named(name)?;
let margin = words.find_map(|word| word.strip_prefix("margin=")?.parse().ok());
Some(Directive::Align { align, margin })
}
fn align_named(name: &str) -> Option<Align> {
match name {
"left" => Some(Align::Left),
"right" => Some(Align::Right),
// Both spellings, folded onto one value.
"center" | "centre" => Some(Align::Center),
_ => None,
}
}
/// Art blocks for a paragraph that holds nothing but art references.
///
/// All or nothing: one ordinary image in the paragraph leaves the whole thing a
/// paragraph, so `![photo](pic.png)` is never mistaken for art. Separating the
/// two by extension is what lets art use standard image syntax at all.
fn art(inline: &[Inline], base: &Path, root: &Path) -> Result<Option<Vec<Block>>, Error> {
let mut images = Vec::new();
for item in inline {
match item {
Inline::Image { src, alt, .. } => images.push((src, alt)),
// Whitespace between references, and the line breaks of a paragraph
// holding several, are not content.
Inline::Text(text) if text.trim().is_empty() => {}
Inline::SoftBreak => {}
_ => return Ok(None),
}
}
if images.is_empty() {
return Ok(None);
}
let mut blocks = Vec::with_capacity(images.len());
for (src, alt) in images {
let Some(path) = art_path(src, base, root) else { return Ok(None) };
let text =
fs::read_to_string(&path).map_err(|cause| Error::Io { path: path.clone(), cause })?;
let (align, alt) = split_align(&Doc::plain_text(alt));
let block = Block::Art { alt, lines: text.lines().map(str::to_string).collect() };
blocks.push(match align {
Some(align) => Block::Aligned { align, margin: None, block: Box::new(block) },
None => block,
});
}
Ok(Some(blocks))
}
/// The file an image target names, if it is a plain-text file inside the root.
fn art_path(src: &str, base: &Path, root: &Path) -> Option<PathBuf> {
// A remote target is not a local file, and must not be fetched.
if src.contains("://") || src.starts_with("//") {
return None;
}
let joined = match src.strip_prefix('/') {
Some(rel) => root.join(rel),
None => base.join(src),
};
let path = canonical_within(&joined, root).ok()?;
(mime::media_type(&path) == mime::PLAIN_TEXT).then_some(path)
}
/// Split leading or trailing `:align` tokens out of an art label.
///
/// They are removed from the label because it is the textual fallback a gemtext
/// or HTML client shows, and a presentation token has no business appearing
/// there. An unrecognised `:token` is left alone, since it is just text.
fn split_align(alt: &str) -> (Option<Align>, String) {
let mut align = None;
let mut words = Vec::new();
for word in alt.split_whitespace() {
match word.strip_prefix(':').and_then(align_named) {
Some(found) => align = Some(found),
None => words.push(word),
}
}
(align, words.join(" "))
}
#[cfg(test)]
mod tests {
use super::*;
use crate::parse;
struct Tree(tempfile::TempDir);
impl Tree {
fn new() -> Self {
Tree(tempfile::tempdir().unwrap())
}
fn root(&self) -> PathBuf {
self.0.path().canonicalize().unwrap()
}
fn write(&self, rel: &str, body: &str) {
let path = self.0.path().join(rel);
if let Some(parent) = path.parent() {
fs::create_dir_all(parent).unwrap();
}
fs::write(&path, body).unwrap();
}
/// Parse `page.md` the way the server does.
fn doc(&self, body: &str) -> Result<Doc, Error> {
self.write("page.md", body);
parse::document(&self.root().join("page.md"), &self.root())
}
}
fn para(text: &str) -> Block {
Block::Paragraph(vec![Inline::Text(text.into())])
}
// -- Card dividers -----------------------------------------------------
#[test]
fn a_card_comment_becomes_a_divider() {
let tree = Tree::new();
let doc = tree.doc("Intro.\n\n<!-- card Weather -->\n\nCold.\n").unwrap();
assert_eq!(
doc.blocks,
vec![para("Intro."), Block::CardBreak { title: Some("Weather".into()) }, para("Cold."),]
);
}
#[test]
fn a_card_comment_works_without_blank_lines_around_it() {
let tree = Tree::new();
let doc = tree.doc("Intro.\n<!-- card Weather -->\nCold.\n").unwrap();
assert_eq!(doc.blocks[1], Block::CardBreak { title: Some("Weather".into()) });
}
#[test]
fn a_card_divider_may_be_untitled() {
let tree = Tree::new();
let doc = tree.doc("<!-- card -->\nx\n").unwrap();
assert_eq!(doc.blocks[0], Block::CardBreak { title: None });
}
#[test]
fn a_thematic_break_stays_a_rule() {
// `---` is the untitled divider, and it is already a standard construct.
let tree = Tree::new();
assert_eq!(tree.doc("a\n\n---\n\nb\n").unwrap().blocks[1], Block::Rule);
}
// -- Alignment ---------------------------------------------------------
#[test]
fn an_alignment_comment_wraps_the_following_block() {
let tree = Tree::new();
let doc = tree.doc("<!-- center -->\nCentred.\n\nPlain.\n").unwrap();
assert_eq!(
doc.blocks,
vec![
Block::Aligned {
align: Align::Center,
margin: None,
block: Box::new(para("Centred.")),
},
// It applies to one block only, not to everything after it.
para("Plain."),
]
);
}
#[test]
fn an_alignment_comment_accepts_a_margin() {
let tree = Tree::new();
let doc = tree.doc("<!-- right margin=4 -->\nx\n").unwrap();
assert_eq!(
doc.blocks[0],
Block::Aligned { align: Align::Right, margin: Some(4), block: Box::new(para("x")) }
);
}
#[test]
fn both_spellings_of_centre_are_accepted() {
let tree = Tree::new();
for spelling in ["center", "centre"] {
let doc = tree.doc(&format!("<!-- {spelling} -->\nx\n")).unwrap();
assert!(matches!(doc.blocks[0], Block::Aligned { align: Align::Center, .. }));
}
}
#[test]
fn an_unrecognised_comment_stays_a_comment() {
// Comments have a legitimate non-directive use, so an ordinary one must
// survive untouched.
let tree = Tree::new();
let doc = tree.doc("<!-- TODO: rewrite this -->\nx\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Html(html) if html.contains("TODO")));
assert_eq!(doc.blocks[1], para("x"));
}
#[test]
fn an_inline_comment_is_never_a_directive() {
let tree = Tree::new();
let doc = tree.doc("text <!-- card Nope --> more\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Paragraph(_)));
assert_eq!(doc.blocks.len(), 1);
}
// -- Art ---------------------------------------------------------------
#[test]
fn an_image_of_a_text_file_becomes_verbatim_art() {
let tree = Tree::new();
tree.write("art/dragon.txt", " /\\ \n/__\\\n");
let doc = tree.doc("![Dragon](art/dragon.txt)\n").unwrap();
assert_eq!(
doc.blocks,
vec![Block::Art { alt: "Dragon".into(), lines: vec![" /\\ ".into(), "/__\\".into()] }]
);
}
#[test]
fn an_art_target_may_be_root_absolute() {
// Unambiguous regardless of which file the reference was written in.
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
tree.write("sub/page.md", "![D](/art/dragon.txt)\n");
let doc = parse::document(&tree.root().join("sub/page.md"), &tree.root()).unwrap();
assert!(matches!(&doc.blocks[0], Block::Art { .. }));
}
#[test]
fn an_align_token_is_taken_out_of_the_label() {
// The label is the fallback a gemtext or HTML client shows, so a
// presentation token must not survive into it.
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
let doc = tree.doc("![:center Dragon](art/dragon.txt)\n").unwrap();
assert_eq!(
doc.blocks[0],
Block::Aligned {
align: Align::Center,
margin: None,
block: Box::new(Block::Art { alt: "Dragon".into(), lines: vec!["x".into()] }),
}
);
}
#[test]
fn an_align_token_is_accepted_after_the_label_too() {
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
let doc = tree.doc("![Dragon :right](art/dragon.txt)\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Aligned { align: Align::Right, .. }));
}
#[test]
fn an_unrecognised_colon_token_stays_in_the_label() {
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
let doc = tree.doc("![:odd Dragon](art/dragon.txt)\n").unwrap();
let Block::Art { alt, .. } = &doc.blocks[0] else { panic!("expected art") };
assert_eq!(alt, ":odd Dragon");
}
#[test]
fn an_images_own_token_wins_over_a_preceding_comment() {
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
let doc = tree.doc("<!-- left -->\n![:right D](art/dragon.txt)\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Aligned { align: Align::Right, .. }));
}
#[test]
fn several_art_references_on_one_line_become_several_blocks() {
let tree = Tree::new();
tree.write("a.txt", "A\n");
tree.write("b.txt", "B\n");
let doc = tree.doc("![one](a.txt) ![two](b.txt)\n").unwrap();
assert_eq!(doc.blocks.len(), 2);
assert!(doc.blocks.iter().all(|b| matches!(b, Block::Art { .. })));
}
#[test]
fn an_ordinary_image_is_left_alone() {
let tree = Tree::new();
tree.write("pic.png", "not really a png");
let doc = tree.doc("![photo](pic.png)\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Paragraph(_)));
}
#[test]
fn an_image_among_text_is_left_alone() {
// Art replaces a whole block; inline it stays an image, which is the
// rule the old `#[](){}` syntax enforced by filling the line.
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
let doc = tree.doc("see ![D](art/dragon.txt) here\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Paragraph(_)));
}
#[test]
fn a_mixed_paragraph_is_left_alone_entirely() {
let tree = Tree::new();
tree.write("art/dragon.txt", "x\n");
tree.write("pic.png", "not really a png");
let doc = tree.doc("![D](art/dragon.txt) ![photo](pic.png)\n").unwrap();
assert_eq!(doc.blocks.len(), 1);
assert!(matches!(&doc.blocks[0], Block::Paragraph(_)));
}
#[test]
fn an_art_target_outside_the_root_is_not_read() {
let tree = Tree::new();
let outside = tree.0.path().parent().unwrap().join("itsybitsy-art-escape.txt");
fs::write(&outside, "secret\n").unwrap();
let doc = tree.doc("![x](../itsybitsy-art-escape.txt)\n").unwrap();
// Refused by being left an ordinary image, so nothing outside is read.
assert!(matches!(&doc.blocks[0], Block::Paragraph(_)));
fs::remove_file(outside).unwrap();
}
#[test]
fn a_remote_image_is_never_fetched() {
let tree = Tree::new();
let doc = tree.doc("![x](https://example.com/a.txt)\n").unwrap();
assert!(matches!(&doc.blocks[0], Block::Paragraph(_)));
}
#[test]
fn directives_inside_quotes_and_lists_are_interpreted() {
let tree = Tree::new();
let doc = tree.doc("> <!-- card Inner -->\n\n- <!-- center -->\n x\n").unwrap();
let Block::BlockQuote(inner) = &doc.blocks[0] else { panic!("expected a quote") };
assert_eq!(inner[0], Block::CardBreak { title: Some("Inner".into()) });
let Block::List { items, .. } = &doc.blocks[1] else { panic!("expected a list") };
assert!(matches!(items[0][0], Block::Aligned { .. }));
}
}

View file

@ -44,19 +44,26 @@ pub enum Block {
head: Vec<Vec<Inline>>,
rows: Vec<Vec<Vec<Inline>>>,
},
/// A verbatim block lifted from a file by `#[label](art.txt)`. Already read,
/// because nothing downstream of the parser touches the filesystem.
/// A verbatim block, from an `![alt](art.txt)` image whose target is a text
/// file. Already read, because nothing downstream of the parser opens files.
Art {
block_type: String,
name: String,
align: Option<Align>,
alt: String,
lines: Vec<String>,
},
/// A `{.card Title}` divider. Formats that paginate start a new screen here;
/// every other format emits nothing at all for it.
/// A `<!-- card Title -->` divider. Formats that paginate start a new screen
/// here; every other format emits nothing at all for it.
CardBreak {
title: Option<String>,
},
/// A block carrying presentation set by a `<!-- center -->` style comment, or
/// by an art label's `:align` token. Only the fixed-width text formats can
/// honour it; the rest unwrap and ignore it, which is why it wraps rather
/// than being a field on every block that might want one.
Aligned {
align: Align,
margin: Option<u16>,
block: Box<Block>,
},
/// Raw block HTML. Kept rather than dropped so XHTML-MP can pass it through
/// and the text formats can strip it, instead of the parser deciding.
Html(String),

View file

@ -6,6 +6,7 @@
pub mod cache;
pub mod config;
pub mod directives;
pub mod error;
pub mod ir;
pub mod mime;

View file

@ -10,6 +10,9 @@ use std::path::Path;
/// Returned when the extension is absent or unrecognised.
pub const DEFAULT: &str = "application/octet-stream";
/// Plain text, which is also what marks a file as inlinable ASCII art.
pub const PLAIN_TEXT: &str = "text/plain; charset=utf-8";
/// The media type to serve `path` as, by extension, compared case-insensitively.
pub fn media_type(path: &Path) -> &'static str {
let Some(extension) = path.extension().and_then(|e| e.to_str()) else { return DEFAULT };
@ -34,7 +37,7 @@ const TABLE: &[(&str, &str)] = &[
("ico", "image/vnd.microsoft.icon"),
("avif", "image/avif"),
// Documents and data
("txt", "text/plain; charset=utf-8"),
("txt", PLAIN_TEXT),
("gmi", "text/gemini; charset=utf-8"),
("css", "text/css; charset=utf-8"),
("html", "text/html; charset=utf-8"),
@ -42,7 +45,7 @@ const TABLE: &[(&str, &str)] = &[
("json", "application/json"),
("toml", "application/toml"),
("pdf", "application/pdf"),
("asc", "text/plain; charset=utf-8"),
("asc", PLAIN_TEXT),
// Audio and video
("mp3", "audio/mpeg"),
("ogg", "audio/ogg"),

View file

@ -12,44 +12,22 @@ use std::path::Path;
use pulldown_cmark::{Alignment, CodeBlockKind, Event, Options, Parser, Tag, TagEnd};
use crate::directives;
use crate::error::Error;
use crate::ir::{Align, Block, Doc, Inline};
use crate::preprocess::{self, Segment};
use crate::preprocess;
/// Read a source file, expand its directives, and parse the result.
pub fn document(source: &Path, root: &Path) -> Result<Doc, Error> {
Ok(from_segments(&preprocess::expand(source, root)?))
}
/// Assemble a document from preprocessed segments.
/// Read a source file and parse it, in the three stages the pipeline has.
///
/// A card break or an art block becomes a block directly, never passing through
/// the Markdown parser. That is what keeps a `{.card}` directive from surfacing
/// as literal paragraph text, which is a bug smolweb works around by stripping
/// the line before one of its two parsers sees it.
pub fn from_segments(segments: &[Segment]) -> Doc {
let mut doc = Doc { blocks: Vec::new(), first_h1: None };
for segment in segments {
match segment {
Segment::Markdown(text) => {
let mut parsed = markdown(text);
if doc.first_h1.is_none() {
doc.first_h1 = parsed.first_h1.take();
}
doc.blocks.append(&mut parsed.blocks);
}
Segment::Art { block_type, name, align, lines } => doc.blocks.push(Block::Art {
block_type: block_type.clone(),
name: name.clone(),
align: *align,
lines: lines.clone(),
}),
Segment::CardBreak { title } => {
doc.blocks.push(Block::CardBreak { title: title.clone() })
}
}
}
doc
/// Includes are spliced first, because they change the token stream; the result
/// is parsed; then the constructs that are standard Markdown — comment
/// directives and art references — are interpreted over the parsed blocks.
/// `source` is expected to be canonical, as [`crate::site::Site`] passes it.
pub fn document(source: &Path, root: &Path) -> Result<Doc, Error> {
let expanded = preprocess::expand(source, root)?;
let mut doc = markdown(&expanded);
directives::apply(&mut doc, source.parent().unwrap_or(root), root)?;
Ok(doc)
}
/// Parse one run of Markdown.
@ -595,61 +573,4 @@ mod tests {
};
assert_eq!(inline[1], Inline::Html("<b>".into()));
}
// -- Segment assembly --------------------------------------------------
#[test]
fn card_breaks_and_art_become_blocks_without_passing_through_the_parser() {
let segments = vec![
Segment::Markdown("Intro.\n".into()),
Segment::CardBreak { title: Some("Weather".into()) },
Segment::Markdown("Cold.\n".into()),
Segment::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec!["/\\".into()],
},
];
let doc = from_segments(&segments);
assert_eq!(
doc.blocks,
vec![
Block::Paragraph(text("Intro.")),
Block::CardBreak { title: Some("Weather".into()) },
Block::Paragraph(text("Cold.")),
Block::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec!["/\\".into()],
},
]
);
}
#[test]
fn a_card_directive_never_appears_as_text() {
// The leak smolweb works around by stripping the line before md2txt sees
// it: here the directive cannot reach the Markdown parser at all.
let dir = tempfile::tempdir().unwrap();
let root = dir.path().canonicalize().unwrap();
std::fs::write(root.join("page.md"), "Intro.\n\n{.card Weather}\nCold.\n").unwrap();
let doc = document(&root.join("page.md"), &root).unwrap();
let rendered = format!("{:?}", doc.blocks);
assert!(!rendered.contains("{.card"), "{rendered}");
assert!(doc.blocks.contains(&Block::CardBreak { title: Some("Weather".into()) }));
}
#[test]
fn the_title_comes_from_the_first_markdown_run_that_has_one() {
let segments = vec![
Segment::CardBreak { title: None },
Segment::Markdown("## not it\n".into()),
Segment::Markdown("# Found\n".into()),
Segment::Markdown("# Later\n".into()),
];
assert_eq!(from_segments(&segments).first_h1.as_deref(), Some("Found"));
}
}

View file

@ -1,22 +1,22 @@
//! The line-level pass that runs before Markdown parsing.
//! Include expansion, the one transformation that has to happen before parsing.
//!
//! Four directives live outside CommonMark, and all four occupy a whole line, so
//! they are recognised here rather than by extending the Markdown parser. That
//! is also how the Python does it; the difference is that this pass emits typed
//! segments instead of encoding them as sentinel strings inside the line stream.
//! `![[path]]` splices another file's lines in place, changing the token stream
//! the Markdown parser sees, so it cannot be handled afterwards. Everything else
//! itsybitsy adds to CommonMark is a construct the parser already understands —
//! an HTML comment or an image — and is interpreted after parsing, by
//! [`crate::directives`].
//!
//! This module is the only part of the rendering pipeline that opens files.
//! Keeping it so means the root-containment check has exactly one home, which is
//! what closes the traversal smolweb has: md2txt resolves an include target and
//! checks only that it exists, so `{.include ../../../../etc/passwd}` in any
//! served document reads and emits that file.
//! Together with that module this is the only part of the pipeline that opens
//! files, which gives the root-containment check exactly one home. That closes
//! the traversal smolweb has: md2txt resolves an include target and checks only
//! that it exists, so `{.include ../../../../etc/passwd}` in any served document
//! reads and emits that file.
use std::collections::BTreeSet;
use std::fs;
use std::path::{Path, PathBuf};
use crate::error::{Error, IncludeReason};
use crate::ir::Align;
/// How deep includes may nest. The cycle set alone does not bound a long chain
/// that never repeats a file.
@ -27,65 +27,35 @@ const MAX_DEPTH: usize = 16;
const MAX_LINES: usize = 200_000;
const MAX_BYTES: usize = 8 * 1024 * 1024;
/// A run of input, classified. Markdown runs are parsed; the rest become blocks
/// directly, so no directive can ever be mistaken for paragraph text.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Segment {
Markdown(String),
Art { block_type: String, name: String, align: Option<Align>, lines: Vec<String> },
CardBreak { title: Option<String> },
}
/// Read `source` and expand it into segments.
/// Read `source` and splice in every file it includes, recursively.
///
/// `root` bounds every path this may read: an include or art target that
/// canonicalises outside it is refused, not followed.
pub fn expand(source: &Path, root: &Path) -> Result<Vec<Segment>, Error> {
let mut out = Segments::default();
/// `root` bounds every path this may read: a target that canonicalises outside
/// it is refused, not followed.
pub fn expand(source: &Path, root: &Path) -> Result<String, Error> {
let mut out = Output::default();
let mut stack = BTreeSet::new();
let canonical = canonical_within(source, root)?;
stack.insert(canonical.clone());
expand_file(&canonical, root, &mut stack, 0, &mut out)?;
Ok(out.finish())
Ok(out.text)
}
/// Accumulates segments, merging consecutive Markdown lines into one run so the
/// Markdown parser sees whole constructs rather than line fragments.
#[derive(Default)]
struct Segments {
done: Vec<Segment>,
markdown: String,
struct Output {
text: String,
lines: usize,
bytes: usize,
}
impl Segments {
impl Output {
fn push_line(&mut self, line: &str) -> Result<(), Error> {
self.lines += 1;
self.bytes += line.len();
if self.lines > MAX_LINES || self.bytes > MAX_BYTES {
if self.lines > MAX_LINES || self.text.len() + line.len() > MAX_BYTES {
return Err(Error::Include { path: PathBuf::new(), reason: IncludeReason::TooLarge });
}
self.markdown.push_str(line);
self.markdown.push('\n');
self.text.push_str(line);
self.text.push('\n');
Ok(())
}
fn push(&mut self, segment: Segment) {
self.flush();
self.done.push(segment);
}
fn flush(&mut self) {
if !self.markdown.is_empty() {
self.done.push(Segment::Markdown(std::mem::take(&mut self.markdown)));
}
}
fn finish(mut self) -> Vec<Segment> {
self.flush();
self.done
}
}
fn expand_file(
@ -93,59 +63,35 @@ fn expand_file(
root: &Path,
stack: &mut BTreeSet<PathBuf>,
depth: usize,
out: &mut Segments,
out: &mut Output,
) -> Result<(), Error> {
let text =
fs::read_to_string(path).map_err(|cause| Error::Io { path: path.to_path_buf(), cause })?;
let base = path.parent().unwrap_or(root);
for line in text.lines() {
if let Some(title) = card_break(line) {
out.push(Segment::CardBreak { title: title.map(str::to_string) });
let Some(target) = include_target(line) else {
out.push_line(line)?;
continue;
};
if depth + 1 > MAX_DEPTH {
return Err(Error::Include {
path: path.to_path_buf(),
reason: IncludeReason::TooDeep,
});
}
if let Some(pieces) = art_pieces(line) {
for (label, target) in pieces {
let (block_type, name, align) = parse_label(label);
let file = resolve(target, base, root)?;
let art = fs::read_to_string(&file)
.map_err(|cause| Error::Io { path: file.clone(), cause })?;
out.push(Segment::Art {
block_type,
name,
align,
lines: art.lines().map(str::to_string).collect(),
});
}
continue;
let file = canonical_within(&base.join(target), root)?;
if !stack.insert(file.clone()) {
return Err(Error::Include { path: file, reason: IncludeReason::Circular });
}
if let Some(target) = include_target(line) {
if depth + 1 > MAX_DEPTH {
return Err(Error::Include {
path: path.to_path_buf(),
reason: IncludeReason::TooDeep,
});
}
let file = resolve(target, base, root)?;
if !stack.insert(file.clone()) {
return Err(Error::Include { path: file, reason: IncludeReason::Circular });
}
expand_file(&file, root, stack, depth + 1, out)?;
stack.remove(&file);
continue;
}
out.push_line(line)?;
expand_file(&file, root, stack, depth + 1, out)?;
stack.remove(&file);
}
Ok(())
}
/// Resolve a directive's target relative to the including file, and require the
/// result to be a readable file inside the content root.
fn resolve(target: &str, base: &Path, root: &Path) -> Result<PathBuf, Error> {
canonical_within(&base.join(target), root)
}
fn canonical_within(path: &Path, root: &Path) -> Result<PathBuf, Error> {
/// Canonicalise a path and require it to be a regular file inside `root`.
pub(crate) fn canonical_within(path: &Path, root: &Path) -> Result<PathBuf, Error> {
let canonical = path
.canonicalize()
.map_err(|_| Error::Include { path: path.to_path_buf(), reason: IncludeReason::Missing })?;
@ -155,204 +101,38 @@ fn canonical_within(path: &Path, root: &Path) -> Result<PathBuf, Error> {
Ok(canonical)
}
// -- Directive recognition ------------------------------------------------
//
// Hand-written rather than regex-driven: each pattern is an anchored prefix and
// suffix around one capture, and a dependency for five of those is not a trade
// worth making in a codebase meant to stay auditable.
/// `{.card}` or `{.card Title}` alone on a line. The outer `Option` is whether
/// the line is a card break at all; the inner one is whether it carries a title.
pub fn card_break(line: &str) -> Option<Option<&str>> {
let rest = line.trim().strip_prefix('{')?.trim_start().strip_prefix(".card")?;
let inner = rest.trim_end().strip_suffix('}')?;
if inner.is_empty() {
return Some(None);
}
// A following character must be whitespace, or this is `{.cardsomething}`.
if !inner.starts_with(char::is_whitespace) {
return None;
}
let title = inner.trim();
Some((!title.is_empty()).then_some(title))
}
/// `![[target]]` or `{.include target}` alone on a line.
/// `![[target]]` alone on a line.
///
/// CommonMark renders this as literal text, which is exactly why it needs a
/// pre-pass; it is also the spelling Obsidian and its relatives established, so
/// a document carrying it stays recognisable elsewhere.
pub fn include_target(line: &str) -> Option<&str> {
let trimmed = line.trim();
if let Some(inner) = trimmed.strip_prefix("![[").and_then(|r| r.strip_suffix("]]")) {
return unquote(inner);
}
let rest = trimmed.strip_prefix('{')?.trim_start().strip_prefix(".include")?;
if !rest.starts_with(char::is_whitespace) {
return None;
}
unquote(rest.trim_end().strip_suffix('}')?)
}
/// Strip surrounding matched quotes, as the Python's target normalisation does.
fn unquote(value: &str) -> Option<&str> {
let trimmed = value.trim();
let inner = match (trimmed.chars().next(), trimmed.chars().last()) {
(Some(first @ ('\'' | '"')), Some(last)) if first == last && trimmed.len() >= 2 => {
trimmed[1..trimmed.len() - 1].trim()
}
_ => trimmed,
};
(!inner.is_empty()).then_some(inner)
}
/// `#[label](target)` pieces filling a whole line, as `(label, target)` pairs.
///
/// One piece is the block form, several the inline form; the Python accepts the
/// inline form only when nothing but whitespace surrounds the pieces, which is
/// what makes this a block-level directive and keeps it from splitting a
/// sentence.
pub fn art_pieces(line: &str) -> Option<Vec<(&str, &str)>> {
let mut rest = line.trim();
// The block form allows a trailing MultiMarkdown attribute list. It is
// recognised so the line is still treated as art, but no format reads it.
if let Some(open) = rest.rfind("{:")
&& rest.ends_with('}')
{
rest = rest[..open].trim_end();
}
let mut pieces = Vec::new();
while !rest.is_empty() {
let (label, target, consumed) = one_art(rest)?;
pieces.push((label, target));
rest = rest[consumed..].trim_start();
}
(!pieces.is_empty()).then_some(pieces)
}
/// One `#[label](target)` at the start of `text`, with the length it consumed.
fn one_art(text: &str) -> Option<(&str, &str, usize)> {
let after_open = text.strip_prefix("#[")?;
let label_end = after_open.find(']')?;
let label = &after_open[..label_end];
let after_label = &after_open[label_end + 1..];
let after_paren = after_label.strip_prefix('(')?;
let target_end = after_paren.find(')')?;
let target = &after_paren[..target_end];
if label.is_empty() || target.is_empty() {
return None;
}
// "#[" + label + "](" + target + ")"
let consumed = 2 + label.len() + 2 + target.len() + 1;
Some((label, target, consumed))
}
/// Split an art label into its block type, name and alignment.
///
/// Tokens beginning with a colon set the alignment; the first remaining token is
/// the block type and the rest its name. An absent type is `custom`, matching
/// the Python.
fn parse_label(label: &str) -> (String, String, Option<Align>) {
let mut align = None;
let mut words = Vec::new();
for token in label.split_whitespace() {
match token.strip_prefix(':') {
Some(tag) => {
align = match tag.to_ascii_lowercase().as_str() {
"left" => Some(Align::Left),
"right" => Some(Align::Right),
"center" | "centre" => Some(Align::Center),
// An unrecognised tag is ignored rather than treated as a name.
_ => align,
};
}
None => words.push(token),
}
}
let block_type = words.first().copied().unwrap_or("custom").to_string();
let name = words.get(1..).map(|rest| rest.join(" ")).unwrap_or_default();
(block_type, name, align)
let inner = line.trim().strip_prefix("![[")?.strip_suffix("]]")?;
let trimmed = inner.trim();
(!trimmed.is_empty()).then_some(trimmed)
}
#[cfg(test)]
mod tests {
use super::*;
// -- Directive recognition ---------------------------------------------
#[test]
fn recognises_card_breaks() {
assert_eq!(card_break("{.card}"), Some(None));
assert_eq!(card_break(" {.card} "), Some(None));
assert_eq!(card_break("{ .card }"), Some(None));
assert_eq!(card_break("{.card Weather}"), Some(Some("Weather")));
assert_eq!(card_break("{ .card Two Words }"), Some(Some("Two Words")));
}
#[test]
fn rejects_near_misses_for_card_breaks() {
assert_eq!(card_break("{.cardinal}"), None);
assert_eq!(card_break("text {.card}"), None);
assert_eq!(card_break("{.card"), None);
assert_eq!(card_break("{.cards Weather}"), None);
}
#[test]
fn recognises_includes_in_both_spellings() {
fn recognises_a_wikilink_include() {
assert_eq!(include_target("![[notes.md]]"), Some("notes.md"));
assert_eq!(include_target(" ![[a/b.md]] "), Some("a/b.md"));
assert_eq!(include_target("{.include notes.md}"), Some("notes.md"));
assert_eq!(include_target("{ .include notes.md }"), Some("notes.md"));
// Quoted targets, as the Python's normalisation allows.
assert_eq!(include_target("{.include \"a b.md\"}"), Some("a b.md"));
assert_eq!(include_target("![['q.md']]"), Some("q.md"));
assert_eq!(include_target("![[ spaced.md ]]"), Some("spaced.md"));
}
#[test]
fn rejects_near_misses_for_includes() {
fn rejects_near_misses() {
// Not alone on its line, so it is ordinary text.
assert_eq!(include_target("see ![[notes.md]] there"), None);
assert_eq!(include_target("{.includes notes.md}"), None);
assert_eq!(include_target("{.include}"), None);
assert_eq!(include_target("![[]]"), None);
assert_eq!(include_target("![[unterminated"), None);
// The directive smolweb also accepted is gone: one spelling, not two.
assert_eq!(include_target("{.include notes.md}"), None);
}
#[test]
fn recognises_art_in_block_and_inline_forms() {
assert_eq!(art_pieces("#[banner](logo.txt)"), Some(vec![("banner", "logo.txt")]));
assert_eq!(
art_pieces("#[a](one.txt) #[b](two.txt)"),
Some(vec![("a", "one.txt"), ("b", "two.txt")])
);
// A trailing attribute list keeps the line recognised as art.
assert_eq!(art_pieces("#[banner](logo.txt){: .wide}"), Some(vec![("banner", "logo.txt")]));
}
#[test]
fn art_must_fill_the_whole_line() {
// Otherwise it would split a sentence, which is why the Python requires
// the surrounding text to be whitespace only.
assert_eq!(art_pieces("see #[a](one.txt)"), None);
assert_eq!(art_pieces("#[a](one.txt) trailing"), None);
assert_eq!(art_pieces("plain text"), None);
assert_eq!(art_pieces("#[a]()"), None);
}
#[test]
fn parses_art_labels() {
assert_eq!(parse_label("banner"), ("banner".into(), String::new(), None));
assert_eq!(parse_label("figlet Big Title"), ("figlet".into(), "Big Title".into(), None));
assert_eq!(
parse_label(":center banner Logo"),
("banner".into(), "Logo".into(), Some(Align::Center))
);
// The British spelling is accepted and folded onto the same value.
assert_eq!(parse_label(":centre x").2, Some(Align::Center));
assert_eq!(parse_label(":LEFT x").2, Some(Align::Left));
// No type at all falls back to `custom`, as the Python does.
assert_eq!(parse_label(":right"), ("custom".into(), String::new(), Some(Align::Right)));
// An unknown colon tag is dropped, not taken as the type.
assert_eq!(parse_label(":nonsense banner"), ("banner".into(), String::new(), None));
}
// -- Expansion ---------------------------------------------------------
struct Tree(tempfile::TempDir);
impl Tree {
@ -373,95 +153,40 @@ mod tests {
path
}
fn expand(&self, rel: &str) -> Result<Vec<Segment>, Error> {
fn expand(&self, rel: &str) -> Result<String, Error> {
expand(&self.0.path().join(rel), &self.root())
}
}
#[test]
fn a_file_with_no_directives_is_one_markdown_run() {
fn a_file_with_no_includes_comes_back_unchanged() {
let tree = Tree::new();
tree.write("page.md", "# Title\n\nBody.\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Markdown("# Title\n\nBody.\n".into())]
);
}
#[test]
fn card_breaks_split_the_markdown_runs() {
let tree = Tree::new();
tree.write("page.md", "Intro.\n\n{.card Weather}\nCold.\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![
Segment::Markdown("Intro.\n\n".into()),
Segment::CardBreak { title: Some("Weather".into()) },
Segment::Markdown("Cold.\n".into()),
]
);
assert_eq!(tree.expand("page.md").unwrap(), "# Title\n\nBody.\n");
}
#[test]
fn an_include_splices_the_targets_lines() {
let tree = Tree::new();
tree.write("part.md", "Shared text.\n");
tree.write("page.md", "Before.\n\n{.include part.md}\n\nAfter.\n");
// The spliced lines join the surrounding run: an include is a text
// substitution, not a structural boundary.
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Markdown("Before.\n\nShared text.\n\nAfter.\n".into())]
);
tree.write("page.md", "Before.\n\n![[part.md]]\n\nAfter.\n");
assert_eq!(tree.expand("page.md").unwrap(), "Before.\n\nShared text.\n\nAfter.\n");
}
#[test]
fn a_wikilink_include_resolves_relative_to_the_including_file() {
fn an_include_resolves_relative_to_the_including_file() {
let tree = Tree::new();
tree.write("sub/part.md", "Nested shared.\n");
tree.write("sub/page.md", "![[part.md]]\n");
assert_eq!(
tree.expand("sub/page.md").unwrap(),
vec![Segment::Markdown("Nested shared.\n".into())]
);
}
#[test]
fn art_is_read_into_the_segment() {
let tree = Tree::new();
tree.write("logo.txt", " /\\ \n/__\\\n");
tree.write("page.md", "#[:center banner Logo](logo.txt)\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Art {
block_type: "banner".into(),
name: "Logo".into(),
align: Some(Align::Center),
lines: vec![" /\\ ".into(), "/__\\".into()],
}]
);
assert_eq!(tree.expand("sub/page.md").unwrap(), "Nested shared.\n");
}
#[test]
fn an_include_outside_the_root_is_refused() {
// The traversal smolweb has: md2txt resolves the target and checks only
// that it exists, with no containment check at all.
let tree = Tree::new();
let outside = tree.0.path().parent().unwrap().join("itsybitsy-escape.md");
fs::write(&outside, "secret\n").unwrap();
tree.write("page.md", "{.include ../itsybitsy-escape.md}\n");
let err = tree.expand("page.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Outside, .. }), "{err}");
fs::remove_file(outside).unwrap();
}
#[test]
fn an_art_target_outside_the_root_is_refused() {
let tree = Tree::new();
let outside = tree.0.path().parent().unwrap().join("itsybitsy-art-escape.txt");
fs::write(&outside, "secret\n").unwrap();
tree.write("page.md", "#[banner x](../itsybitsy-art-escape.txt)\n");
tree.write("page.md", "![[../itsybitsy-escape.md]]\n");
let err = tree.expand("page.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Outside, .. }), "{err}");
@ -471,7 +196,7 @@ mod tests {
#[test]
fn a_missing_target_is_reported_not_ignored() {
let tree = Tree::new();
tree.write("page.md", "{.include absent.md}\n");
tree.write("page.md", "![[absent.md]]\n");
let err = tree.expand("page.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Missing, .. }), "{err}");
}
@ -479,8 +204,8 @@ mod tests {
#[test]
fn a_circular_include_is_refused() {
let tree = Tree::new();
tree.write("a.md", "A\n{.include b.md}\n");
tree.write("b.md", "B\n{.include a.md}\n");
tree.write("a.md", "A\n![[b.md]]\n");
tree.write("b.md", "B\n![[a.md]]\n");
let err = tree.expand("a.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Circular, .. }), "{err}");
}
@ -488,7 +213,7 @@ mod tests {
#[test]
fn a_self_include_is_refused() {
let tree = Tree::new();
tree.write("a.md", "{.include a.md}\n");
tree.write("a.md", "![[a.md]]\n");
let err = tree.expand("a.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::Circular, .. }), "{err}");
}
@ -499,7 +224,7 @@ mod tests {
// cap is for.
let tree = Tree::new();
for i in 0..=MAX_DEPTH + 1 {
tree.write(&format!("n{i}.md"), &format!("line {i}\n{{.include n{}.md}}\n", i + 1));
tree.write(&format!("n{i}.md"), &format!("line {i}\n![[n{}.md]]\n", i + 1));
}
let err = tree.expand("n0.md").unwrap_err();
assert!(matches!(err, Error::Include { reason: IncludeReason::TooDeep, .. }), "{err}");
@ -515,7 +240,7 @@ mod tests {
let mut previous = "leaf.md".to_string();
for level in 0..MAX_DEPTH - 2 {
let name = format!("l{level}.md");
tree.write(&name, &format!("{{.include {previous}}}\n{{.include {previous}}}\n"));
tree.write(&name, &format!("![[{previous}]]\n![[{previous}]]\n"));
previous = name;
}
let err = tree.expand(&previous).unwrap_err();
@ -528,10 +253,7 @@ mod tests {
// sequence is fine and must not be mistaken for a cycle.
let tree = Tree::new();
tree.write("part.md", "shared\n");
tree.write("page.md", "{.include part.md}\n{.include part.md}\n");
assert_eq!(
tree.expand("page.md").unwrap(),
vec![Segment::Markdown("shared\nshared\n".into())]
);
tree.write("page.md", "![[part.md]]\n![[part.md]]\n");
assert_eq!(tree.expand("page.md").unwrap(), "shared\nshared\n");
}
}