diff --git a/README.md b/README.md index 76fbb11..9c54d82 100644 --- a/README.md +++ b/README.md @@ -8,12 +8,13 @@ A Rust re-implementation of [smolweb](https://code.randogoth.com/randogoth/smolw ## Status -Early. The configuration layer is implemented and tested; nothing is served yet. +Early. Configuration and path resolution are implemented and tested; nothing is served yet. | Area | State | | --- | --- | | Server configuration, virtual hosts, `--check` | implemented | -| Path resolution and the render cache | planned | +| Path resolution, traversal containment, live reload | implemented | +| Markdown parsing | planned | | gemtext and HTML output | planned | | HTTP, Spartan and Nex listeners | planned | | 80-column text and Nex output | planned | @@ -40,6 +41,19 @@ default_site = "smol" A `.itsybitsy.toml` in any content directory says how the Markdown in **that directory** renders. There is no inheritance: a subdirectory without its own file uses the built-in defaults rather than its parent's. Repetition in a deep tree is the price of never having to look elsewhere to know how a page renders. +```toml +[defaults] +cache_control = 3600 + +[page."index.md"] +title = "Notes" +cache_control = 300 +``` + +`[defaults]` applies to every Markdown file in the directory; a `[page."name.md"]` table overrides it for one file. `title` is per-page by nature, so setting it in `[defaults]` is an error; left unset, it is derived from the page's first level-1 heading. Each renderer's own keys arrive with that renderer. Editing the file re-renders that directory's pages on the next request, and a file that fails to parse makes them error rather than silently falling back to defaults. + +Because the name begins with a dot, the same rule that keeps `.secret.md` out of the URL space keeps this file out of it too. + `--check` validates the configuration and reports the routing it resolved, without binding a port: ```console @@ -56,6 +70,19 @@ formats rendered per page: html, xhtmlmp Everything the schema cannot express is checked here rather than on first request: a format no enabled feature provides, two sites claiming one hostname, a content root that does not exist, a server file sitting inside a content root where it would be served. +## URLs + +Links should be root-relative and extensionless (`[about](/about)`, not `about.md`), so the same link resolves identically from every protocol. + +| Request | Serves | +| --- | --- | +| `/` | `index.md` | +| `/foo` | `foo.md`, else `foo/index.md` | +| `/foo.md` | redirects to `/foo` | +| `/img.png` | the file itself, by media type | + +Nothing outside the content root is reachable. A request target is percent-decoded before it is normalised, so an encoded `..` becomes a real one and gets clamped at the root rather than quietly matching nothing; any path component beginning with a dot is refused outright; and whatever survives is canonicalised and required to still be inside the root, which is what defeats a symlink pointing out of it. There are no generated directory listings — only `index.md`. + ## Development ```bash diff --git a/bin/src/lib.rs b/bin/src/lib.rs index bcd59d9..2ce152c 100644 --- a/bin/src/lib.rs +++ b/bin/src/lib.rs @@ -11,6 +11,7 @@ use std::path::Path; use anyhow::{Context, Result}; use itsybitsy_core::config::{Checked, ServerConfig}; +use itsybitsy_core::siteset::SiteSet; use crate::cli::Cli; @@ -28,24 +29,28 @@ pub fn run(cli: &Cli) -> Result<()> { /// Takes the format set rather than reading [`registry`] directly, so a test can /// supply the ids a later milestone's registry will provide. pub fn check(config_path: &Path, formats: &[&str]) -> Result<()> { - let (config, checked) = load(config_path, formats)?; - report(&config, &checked); + let (config, checked, sites) = load(config_path, formats)?; + report(&config, &checked, &sites); Ok(()) } -fn load(config_path: &Path, formats: &[&str]) -> Result<(ServerConfig, Checked)> { +fn load(config_path: &Path, formats: &[&str]) -> Result<(ServerConfig, Checked, SiteSet)> { // No context on load: its errors already name the file. let config = ServerConfig::load(config_path)?; let checked = config .validate(config_path, formats) .with_context(|| format!("validating {}", config_path.display()))?; - Ok((config, checked)) + // Opening every root here means a broken site fails at startup rather than + // on the first request that happens to name it. + let sites = SiteSet::build(&config) + .with_context(|| format!("opening the sites in {}", config_path.display()))?; + Ok((config, checked, sites)) } /// What `--check` prints: the resolved routing, so an operator can see which /// name reaches which folder without sending a request. -fn report(config: &ServerConfig, checked: &Checked) { - println!("sites:"); +fn report(config: &ServerConfig, checked: &Checked, sites: &SiteSet) { + println!("sites: {} configured, {} distinct roots", checked.roots.len(), sites.len()); for (name, root) in &checked.roots { println!(" {name}: {}", root.display()); } diff --git a/core/src/cache.rs b/core/src/cache.rs new file mode 100644 index 0000000..e24d4ae --- /dev/null +++ b/core/src/cache.rs @@ -0,0 +1,178 @@ +//! Deriving a value from a file once, and again when the file changes. +//! +//! Invalidation is by modification time and length, which is what makes editing +//! a file enough to see the change on the next request with no watcher and no +//! restart. Values are built outside the lock, so two first hits on one file can +//! both build it; the work is idempotent and the second insert wins, which is +//! the trade the Python makes too. + +use std::collections::HashMap; +use std::fs; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, RwLock}; +use std::time::SystemTime; + +use crate::error::Error; + +/// The file state a cached value was derived from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Stamp { + modified: SystemTime, + len: u64, +} + +impl Stamp { + /// Stat `path`. Fails if it cannot be read, which is also how a deleted file + /// invalidates its entry. + pub fn of(path: &Path) -> Result { + let meta = + fs::metadata(path).map_err(|cause| Error::Io { path: path.to_path_buf(), cause })?; + let modified = + meta.modified().map_err(|cause| Error::Io { path: path.to_path_buf(), cause })?; + Ok(Stamp { modified, len: meta.len() }) + } +} + +/// A map from file path to a value derived from that file. +pub struct StatCache { + entries: RwLock)>>, +} + +impl Default for StatCache { + fn default() -> Self { + Self::new() + } +} + +impl StatCache { + pub fn new() -> Self { + StatCache { entries: RwLock::new(HashMap::new()) } + } + + /// The cached value for `path`, or `build`'s result, stored for next time. + /// + /// `build` is given the stamp the value will be recorded against, so a + /// caller that depends on a second file (a page on its directory's config) + /// can combine stamps without stating the file twice. + pub fn get_or_insert_with( + &self, + path: &Path, + build: impl FnOnce() -> Result, + ) -> Result, Error> { + let stamp = Stamp::of(path)?; + if let Some(value) = self.get(path, &stamp) { + return Ok(value); + } + let value = Arc::new(build()?); + self.entries + .write() + .expect("cache lock") + .insert(path.to_path_buf(), (stamp, value.clone())); + Ok(value) + } + + /// The cached value, if one was derived from exactly this file state. + pub fn get(&self, path: &Path, stamp: &Stamp) -> Option> { + let entries = self.entries.read().expect("cache lock"); + let (cached, value) = entries.get(path)?; + (cached == stamp).then(|| value.clone()) + } + + pub fn len(&self) -> usize { + self.entries.read().expect("cache lock").len() + } + + pub fn is_empty(&self) -> bool { + self.len() == 0 + } +} + +#[cfg(test)] +mod tests { + use std::fs::File; + use std::sync::atomic::{AtomicUsize, Ordering}; + use std::time::Duration; + + use super::*; + + /// Force a modification time change. Filesystem granularity is coarse + /// enough that two writes in one test can otherwise share a timestamp, + /// which is why the Python's live-reload test does the same thing. + fn bump_mtime(path: &Path) { + let stamp = Stamp::of(path).unwrap(); + let later = stamp.modified + Duration::from_secs(5); + File::options() + .write(true) + .open(path) + .unwrap() + .set_times(fs::FileTimes::new().set_accessed(later).set_modified(later)) + .unwrap(); + } + + #[test] + fn an_unchanged_file_reuses_the_cached_value() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("page.md"); + fs::write(&path, "first").unwrap(); + let cache: StatCache = StatCache::new(); + let builds = AtomicUsize::new(0); + + let build = || { + builds.fetch_add(1, Ordering::Relaxed); + Ok(fs::read_to_string(&path).unwrap()) + }; + let first = cache.get_or_insert_with(&path, build).unwrap(); + let second = cache.get_or_insert_with(&path, build).unwrap(); + + // Identity, not just equality: the Python asserts `first is second`. + assert!(Arc::ptr_eq(&first, &second)); + assert_eq!(builds.load(Ordering::Relaxed), 1); + } + + #[test] + fn an_edited_file_is_rebuilt() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("page.md"); + fs::write(&path, "first").unwrap(); + let cache: StatCache = StatCache::new(); + + let build = || Ok(fs::read_to_string(&path).unwrap()); + assert_eq!(*cache.get_or_insert_with(&path, build).unwrap(), "first"); + + fs::write(&path, "second").unwrap(); + bump_mtime(&path); + + assert_eq!(*cache.get_or_insert_with(&path, build).unwrap(), "second"); + assert_eq!(cache.len(), 1, "the entry is replaced, not added to"); + } + + #[test] + fn a_same_length_edit_is_still_rebuilt() { + // Length alone would miss this; the modification time catches it. + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("page.md"); + fs::write(&path, "aaaaa").unwrap(); + let cache: StatCache = StatCache::new(); + + let build = || Ok(fs::read_to_string(&path).unwrap()); + assert_eq!(*cache.get_or_insert_with(&path, build).unwrap(), "aaaaa"); + + fs::write(&path, "bbbbb").unwrap(); + bump_mtime(&path); + + assert_eq!(*cache.get_or_insert_with(&path, build).unwrap(), "bbbbb"); + } + + #[test] + fn a_missing_file_is_an_error_not_a_stale_value() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("page.md"); + fs::write(&path, "here").unwrap(); + let cache: StatCache = StatCache::new(); + let build = || Ok(fs::read_to_string(&path).unwrap_or_default()); + cache.get_or_insert_with(&path, build).unwrap(); + + fs::remove_file(&path).unwrap(); + assert!(matches!(cache.get_or_insert_with(&path, build), Err(Error::Io { .. }))); + } +} diff --git a/core/src/config.rs b/core/src/config.rs index fa6d283..ca0df88 100644 --- a/core/src/config.rs +++ b/core/src/config.rs @@ -369,6 +369,97 @@ pub fn normalize_host(raw: &str) -> Option { allowed.then_some(host) } +/// How the Markdown in one directory renders. +/// +/// There is no inheritance. A subdirectory without its own `.itsybitsy.toml` +/// uses the built-in defaults rather than its parent's, so how a page renders is +/// knowable from its own directory alone. Repetition in a deep tree is the price. +#[derive(Debug, Default, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct DirConfig { + /// Applied to every Markdown file in the directory. + #[serde(default)] + pub defaults: PageKeys, + /// Overrides for one file, keyed by its bare name. + #[serde(default)] + pub page: BTreeMap, +} + +/// The deserialisation shape: every key optional, so "unset" is distinguishable +/// from "set to what the default happens to be". +#[derive(Debug, Default, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PageKeys { + pub title: Option, + pub cache_control: Option, +} + +/// Settings resolved for one page: built-in defaults, then `[defaults]`, then +/// the file's own `[page."...'"]` table. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct PageSettings { + /// An explicitly configured title. Absent means derive one, from the first + /// level-1 heading if there is one and the file stem otherwise. + pub title: Option, + /// `Cache-Control: max-age` in seconds. Absent means send no such header. + pub cache_control: Option, +} + +impl DirConfig { + /// Read and parse a per-directory config. A parse failure is returned rather + /// than swallowed: falling back to defaults after a typo is how a site ships + /// wrong output without anyone noticing. + pub fn load(path: &Path) -> Result { + let text = fs::read_to_string(path) + .map_err(|cause| Error::Io { path: path.to_path_buf(), cause })?; + let config: DirConfig = toml::from_str(&text) + .map_err(|cause| Error::Toml { path: path.to_path_buf(), cause })?; + config.validate(path)?; + Ok(config) + } + + fn validate(&self, path: &Path) -> Result<(), Error> { + if self.defaults.title.is_some() { + return Err(Error::config(format!( + "{}: title belongs to one page, so it cannot be set in [defaults]", + path.display() + ))); + } + for name in self.page.keys() { + // A key naming another directory is what "no inheritance" forbids, + // and this is where that is mechanically enforced. + if name.contains('/') { + return Err(Error::config(format!( + "{}: page key '{name}' must be a bare file name in this directory", + path.display() + ))); + } + // A key that cannot name a Markdown file would silently do nothing. + if !name.ends_with(".md") { + return Err(Error::config(format!( + "{}: page key '{name}' names no Markdown file; it must end in .md", + path.display() + ))); + } + } + Ok(()) + } + + /// Settings for one file in this directory, by its bare name. + pub fn settings_for(&self, file_name: &str) -> PageSettings { + let mut settings = PageSettings { title: None, cache_control: self.defaults.cache_control }; + if let Some(page) = self.page.get(file_name) { + if page.title.is_some() { + settings.title = page.title.clone(); + } + if page.cache_control.is_some() { + settings.cache_control = page.cache_control; + } + } + settings + } +} + #[cfg(test)] mod tests { use super::*; @@ -581,3 +672,71 @@ mod tests { assert_eq!(normalize_host("[::1"), None); } } + +#[cfg(test)] +mod dir_config_tests { + use super::*; + + fn load(body: &str) -> Result { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join(DIR_CONFIG); + fs::write(&path, body).unwrap(); + DirConfig::load(&path) + } + + #[test] + fn an_absent_file_means_built_in_defaults() { + let config = DirConfig::default(); + assert_eq!(config.settings_for("index.md"), PageSettings::default()); + } + + #[test] + fn a_page_table_overrides_the_directory_defaults() { + let config = load( + "[defaults]\ncache_control = 3600\n\n\ + [page.\"index.md\"]\ntitle = \"Notes\"\ncache_control = 300\n", + ) + .unwrap(); + + let index = config.settings_for("index.md"); + assert_eq!(index.title.as_deref(), Some("Notes")); + assert_eq!(index.cache_control, Some(300)); + + // A file with no table of its own still gets the directory defaults, + // and no title, which is the signal to derive one. + let other = config.settings_for("about.md"); + assert_eq!(other.title, None); + assert_eq!(other.cache_control, Some(3600)); + } + + #[test] + fn rejects_an_unknown_key() { + let err = load("[defaults]\ncache_contrl = 60\n").unwrap_err(); + assert!(matches!(err, Error::Toml { .. }), "{err}"); + } + + #[test] + fn rejects_a_title_in_defaults() { + // A directory-wide title would silently give every page the same one. + let err = load("[defaults]\ntitle = \"Everything\"\n").unwrap_err(); + assert!(err.to_string().contains("cannot be set in [defaults]"), "{err}"); + } + + #[test] + fn rejects_a_page_key_naming_another_directory() { + let err = load("[page.\"sub/index.md\"]\ntitle = \"No\"\n").unwrap_err(); + assert!(err.to_string().contains("bare file name"), "{err}"); + } + + #[test] + fn rejects_a_page_key_that_cannot_name_a_markdown_file() { + let err = load("[page.index]\ntitle = \"No\"\n").unwrap_err(); + assert!(err.to_string().contains("must end in .md"), "{err}"); + } + + #[test] + fn a_parse_failure_surfaces_rather_than_falling_back() { + let err = load("[defaults]\ncache_control =\n").unwrap_err(); + assert!(matches!(err, Error::Toml { .. }), "{err}"); + } +} diff --git a/core/src/lib.rs b/core/src/lib.rs index 5307724..4c23113 100644 --- a/core/src/lib.rs +++ b/core/src/lib.rs @@ -4,7 +4,12 @@ //! without binding a port; the binary crate holds the protocol listeners and //! is the only place that names a format crate. +pub mod cache; pub mod config; pub mod error; +pub mod mime; +pub mod path; +pub mod site; +pub mod siteset; pub use error::Error; diff --git a/core/src/mime.rs b/core/src/mime.rs new file mode 100644 index 0000000..534e9fa --- /dev/null +++ b/core/src/mime.rs @@ -0,0 +1,78 @@ +//! Media types for files served byte for byte. +//! +//! A fixed table rather than a system lookup: Python's `mimetypes` consults +//! `/etc/mime.types` where it exists, so smolweb's `Content-Type` for the same +//! file differs between hosts. Determinism is worth more here than coverage of +//! the long tail, and an unknown extension has a correct answer anyway. + +use std::path::Path; + +/// Returned when the extension is absent or unrecognised. +pub const DEFAULT: &str = "application/octet-stream"; + +/// The media type to serve `path` as, by extension, compared case-insensitively. +pub fn media_type(path: &Path) -> &'static str { + let Some(extension) = path.extension().and_then(|e| e.to_str()) else { return DEFAULT }; + let extension = extension.to_ascii_lowercase(); + TABLE + .iter() + .find(|(known, _)| *known == extension) + .map(|(_, media_type)| *media_type) + .unwrap_or(DEFAULT) +} + +/// Extensions a folder of Markdown actually carries. Text types name their +/// charset, since a client has no other way to know it. +const TABLE: &[(&str, &str)] = &[ + // Images + ("png", "image/png"), + ("jpg", "image/jpeg"), + ("jpeg", "image/jpeg"), + ("gif", "image/gif"), + ("webp", "image/webp"), + ("svg", "image/svg+xml"), + ("ico", "image/vnd.microsoft.icon"), + ("avif", "image/avif"), + // Documents and data + ("txt", "text/plain; charset=utf-8"), + ("gmi", "text/gemini; charset=utf-8"), + ("css", "text/css; charset=utf-8"), + ("html", "text/html; charset=utf-8"), + ("xml", "application/xml"), + ("json", "application/json"), + ("toml", "application/toml"), + ("pdf", "application/pdf"), + ("asc", "text/plain; charset=utf-8"), + // Audio and video + ("mp3", "audio/mpeg"), + ("ogg", "audio/ogg"), + ("opus", "audio/ogg"), + ("flac", "audio/flac"), + ("wav", "audio/wav"), + ("mp4", "video/mp4"), + ("webm", "video/webm"), + // Archives and fonts + ("gz", "application/gzip"), + ("zip", "application/zip"), + ("tar", "application/x-tar"), + ("woff2", "font/woff2"), +]; + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn recognises_common_extensions() { + assert_eq!(media_type(Path::new("img.png")), "image/png"); + assert_eq!(media_type(Path::new("a/b/photo.JPEG")), "image/jpeg"); + assert_eq!(media_type(Path::new("notes.txt")), "text/plain; charset=utf-8"); + } + + #[test] + fn falls_back_for_the_unknown_and_the_extensionless() { + assert_eq!(media_type(Path::new("mystery.xyz")), DEFAULT); + assert_eq!(media_type(Path::new("LICENSE")), DEFAULT); + assert_eq!(media_type(Path::new("archive.tar.zst")), DEFAULT); + } +} diff --git a/core/src/path.rs b/core/src/path.rs new file mode 100644 index 0000000..b57d9c2 --- /dev/null +++ b/core/src/path.rs @@ -0,0 +1,197 @@ +//! Turning a request target into a root-relative path. +//! +//! The ordering here is load-bearing and ported deliberately: percent-decode +//! *before* normalising, so an encoded `%2e%2e` becomes a real `..` that the +//! normaliser then clamps at the root rather than a literal segment that +//! quietly matches nothing. Normalisation alone cannot be tricked into a +//! traversal string; the canonicalisation check in `site` is the second, +//! independent layer that guards the filesystem side. + +use std::path::Path; + +/// Normalise a request target into a root-relative path with no leading or +/// trailing slash. The root itself is the empty string. +/// +/// Returns `None` when the path must not be served: any component beginning +/// with a dot. That rule covers `.`/`..` left over from normalisation, dotfiles +/// generally, and with them the per-directory `.itsybitsy.toml`. +pub fn clean_path(url_path: &str) -> Option { + let without_query = url_path.split(['?', '#']).next().unwrap_or(""); + let decoded = percent_decode(without_query); + + // `..` pops, and popping an empty stack is how the root clamps it: the + // Python relies on `posixpath.normpath` of a leading-slash path for this. + let mut parts: Vec<&str> = Vec::new(); + for part in decoded.split('/') { + match part { + "" | "." => {} + ".." => { + parts.pop(); + } + other => parts.push(other), + } + } + + if parts.iter().any(|part| part.starts_with('.')) { + return None; + } + Some(parts.join("/")) +} + +/// The canonical root-relative URL a source file is served at. +/// +/// An `index.md` is addressed by its directory, so it yields `/` or `/dir/`; +/// anything else drops its extension. Note that `/dir` and `/dir/` both resolve +/// to the same page and neither is redirected to the other, which is the +/// behaviour being ported. +pub fn url_for(root: &Path, source: &Path) -> Option { + let rel = source.strip_prefix(root).ok()?; + if rel.file_name()? == "index.md" { + let parent = rel.parent()?.to_str()?; + return Some(if parent.is_empty() { "/".to_string() } else { format!("/{parent}/") }); + } + Some(format!("/{}", rel.with_extension("").to_str()?)) +} + +/// Decode `%XX` escapes, leaving an invalid escape as the literal text it is. +/// +/// Bytes that do not form valid UTF-8 become U+FFFD, which matches no filename, +/// so a malformed target resolves to nothing rather than erroring. That matches +/// Python's lossy `unquote` and is pinned by a test. +fn percent_decode(raw: &str) -> String { + let bytes = raw.as_bytes(); + let mut out = Vec::with_capacity(bytes.len()); + let mut i = 0; + while i < bytes.len() { + if bytes[i] == b'%' + && i + 2 < bytes.len() + && let (Some(hi), Some(lo)) = (hex_digit(bytes[i + 1]), hex_digit(bytes[i + 2])) + { + out.push(hi << 4 | lo); + i += 3; + continue; + } + out.push(bytes[i]); + i += 1; + } + String::from_utf8_lossy(&out).into_owned() +} + +fn hex_digit(byte: u8) -> Option { + match byte { + b'0'..=b'9' => Some(byte - b'0'), + b'a'..=b'f' => Some(byte - b'a' + 10), + b'A'..=b'F' => Some(byte - b'A' + 10), + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::path::PathBuf; + + #[track_caller] + fn clean(path: &str) -> String { + clean_path(path).expect("should resolve") + } + + #[test] + fn root_forms_are_the_empty_path() { + assert_eq!(clean("/"), ""); + assert_eq!(clean(""), ""); + assert_eq!(clean("//"), ""); + } + + #[test] + fn strips_query_and_fragment() { + assert_eq!(clean("/about?format=html"), "about"); + assert_eq!(clean("/about#top"), "about"); + assert_eq!(clean("/about?a=1#top"), "about"); + } + + #[test] + fn collapses_slashes_and_trailing_separators() { + assert_eq!(clean("/dir/"), "dir"); + assert_eq!(clean("//dir//page//"), "dir/page"); + assert_eq!(clean("dir/page"), "dir/page"); + } + + #[test] + fn clamps_traversal_at_the_root() { + // Ported from smolweb's TestPathTraversal: these must never reach above + // the root, and since nothing is mounted at the clamped path they + // resolve to a path that simply does not exist. + assert_eq!(clean("/../../etc/passwd"), "etc/passwd"); + assert_eq!(clean("/../../../../../../etc/passwd"), "etc/passwd"); + assert_eq!(clean("/foo/../../etc/passwd"), "etc/passwd"); + } + + #[test] + fn decodes_before_normalising() { + // The point of the ordering: an encoded `..` has to become a real one + // before the clamp runs, or it would be treated as a literal segment. + assert_eq!(clean("/%2e%2e/etc/passwd"), "etc/passwd"); + assert_eq!(clean("/%2E%2E/etc/passwd"), "etc/passwd"); + // An encoded separator becomes a separator, as Python's unquote does. + assert_eq!(clean("/dir%2fpage"), "dir/page"); + assert_eq!(clean("/hello%20world"), "hello world"); + } + + #[test] + fn rejects_dotfiles() { + assert_eq!(clean_path("/.secret"), None); + assert_eq!(clean_path("/dir/.secret"), None); + // The per-directory config is a dotfile, so this rule hides it. + assert_eq!(clean_path("/.itsybitsy.toml"), None); + assert_eq!(clean_path("/notes/.itsybitsy.toml"), None); + // Surviving dot segments are caught defensively too. + assert_eq!(clean_path("/%2e%2e%2f.secret"), None); + } + + #[test] + fn a_lone_dot_is_the_root_not_a_dotfile() { + assert_eq!(clean("/."), ""); + assert_eq!(clean("/./"), ""); + } + + #[test] + fn an_invalid_escape_stays_literal() { + assert_eq!(clean("/%zz"), "%zz"); + assert_eq!(clean("/%"), "%"); + assert_eq!(clean("/%4"), "%4"); + } + + #[test] + fn invalid_utf8_becomes_a_replacement_character() { + // U+FFFD matches no filename, so this resolves to nothing rather than + // erroring. Pinned so a later cleanup does not turn it into a 400. + assert_eq!(clean("/%ff"), "\u{fffd}"); + } + + #[test] + fn a_backslash_is_an_ordinary_character() { + // posixpath treats it as part of a name, and so must this. + assert_eq!(clean("/dir\\page"), "dir\\page"); + } + + #[test] + fn urls_for_index_files_address_their_directory() { + let root = PathBuf::from("/srv/content"); + assert_eq!(url_for(&root, &root.join("index.md")).unwrap(), "/"); + assert_eq!(url_for(&root, &root.join("dir/index.md")).unwrap(), "/dir/"); + assert_eq!(url_for(&root, &root.join("a/b/index.md")).unwrap(), "/a/b/"); + } + + #[test] + fn urls_for_other_files_drop_the_extension() { + let root = PathBuf::from("/srv/content"); + assert_eq!(url_for(&root, &root.join("about.md")).unwrap(), "/about"); + assert_eq!(url_for(&root, &root.join("dir/page.md")).unwrap(), "/dir/page"); + } + + #[test] + fn url_for_refuses_a_path_outside_the_root() { + assert_eq!(url_for(Path::new("/srv/content"), Path::new("/etc/passwd")), None); + } +} diff --git a/core/src/site.rs b/core/src/site.rs new file mode 100644 index 0000000..1d5d6a5 --- /dev/null +++ b/core/src/site.rs @@ -0,0 +1,450 @@ +//! One content root, and what a request path resolves to inside it. +//! +//! Containment is enforced twice, independently. `path::clean_path` cannot +//! produce a string that escapes the root; `safe_file` then canonicalises what +//! that string names and requires the result to still be under the root, which +//! is what defeats a symlink pointing outside. There is a residual window +//! between canonicalising and opening a file; for an author-controlled tree that +//! is acceptable, and it is the reason the canonical path is what gets opened. + +use std::path::{Path, PathBuf}; +use std::sync::Arc; + +use crate::cache::StatCache; +use crate::config::{DIR_CONFIG, DirConfig, PageSettings}; +use crate::error::Error; +use crate::mime; +use crate::path::{clean_path, url_for}; + +/// How deep a chain of server-side redirects `resolve_flat` will follow. Every +/// redirect currently points at a directly resolvable document, so one hop is +/// always enough; the cap is there so a future rule cannot loop. +const MAX_FLAT_HOPS: usize = 5; + +/// What a request path resolved to. +#[derive(Debug)] +pub enum Resolution { + Found(Resource), + /// The resource lives at this canonical root-relative URL. + Redirect(String), + NotFound, +} + +#[derive(Debug)] +pub enum Resource { + /// A Markdown document, to be rendered into the format the caller serves. + Document { url: String, source: PathBuf, settings: PageSettings }, + /// A file served byte for byte. + Raw { path: PathBuf, media_type: &'static str }, +} + +pub struct Site { + root: PathBuf, + dir_configs: StatCache, + /// Shared by every directory without a config of its own, so the common case + /// allocates nothing. + built_in: Arc, +} + +impl Site { + /// Open a content root, canonicalising it so containment checks have a + /// stable base and a missing root fails now rather than per request. + pub fn new(root: &Path) -> Result { + let root = + root.canonicalize().map_err(|cause| Error::Io { path: root.to_path_buf(), cause })?; + if !root.is_dir() { + return Err(Error::config(format!("{} is not a directory", root.display()))); + } + Ok(Site { root, dir_configs: StatCache::new(), built_in: Arc::new(DirConfig::default()) }) + } + + pub fn root(&self) -> &Path { + &self.root + } + + /// Resolve a request path. + /// + /// | Request | Serves | + /// | --- | --- | + /// | `/` | `index.md` | + /// | `/foo` | `foo.md`, else `foo/index.md` | + /// | `/foo.md` | redirects to `/foo` | + /// | `/img.png` | the file itself, by media type | + pub fn resolve(&self, url_path: &str) -> Result { + let Some(clean) = clean_path(url_path) else { return Ok(Resolution::NotFound) }; + let Some(source) = self.file_for(&clean) else { return Ok(Resolution::NotFound) }; + + // `/foo.md` always redirects to its extensionless form, so one + // root-relative link works identically from every protocol. + if clean.ends_with(".md") { + return Ok(match url_for(&self.root, &source) { + Some(url) => Resolution::Redirect(url), + None => Resolution::NotFound, + }); + } + + if source.extension().and_then(|e| e.to_str()) != Some("md") { + let media_type = mime::media_type(&source); + return Ok(Resolution::Found(Resource::Raw { path: source, media_type })); + } + + let Some(url) = url_for(&self.root, &source) else { return Ok(Resolution::NotFound) }; + let settings = self.settings_for(&source)?; + Ok(Resolution::Found(Resource::Document { url, source, settings })) + } + + /// Resolve the way [`Site::resolve`] does, but never hand back a redirect. + /// + /// Nex and Gopher have no redirect status, so there is nothing to bounce a + /// client with: the canonical target is resolved here instead and its content + /// served directly, on the first request. + pub fn resolve_flat(&self, url_path: &str) -> Result { + let mut target = url_path.to_string(); + for _ in 0..MAX_FLAT_HOPS { + match self.resolve(&target)? { + Resolution::Redirect(location) => target = location, + settled => return Ok(settled), + } + } + Ok(Resolution::NotFound) + } + + /// The source file a cleaned path names, if one exists and is contained. + fn file_for(&self, clean: &str) -> Option { + if clean.is_empty() { + return self.safe_file(&self.root.join("index.md")); + } + + // A name that already carries an extension resolves literally: `.md` is + // never appended to a file that was asked for by name. + let named = Path::new(clean).file_name()?.to_str()?; + if named.contains('.') { + return self.safe_file(&self.root.join(clean)); + } + + self.safe_file(&self.root.join(format!("{clean}.md"))) + .or_else(|| self.safe_file(&self.root.join(clean).join("index.md"))) + } + + /// The canonical path of a file inside the root, or `None`. + /// + /// Canonicalising first and testing `is_file` on the *result* is what stops a + /// symlink to a directory, or to anything outside the root, from passing. + fn safe_file(&self, path: &Path) -> Option { + let resolved = path.canonicalize().ok()?; + (resolved.starts_with(&self.root) && resolved.is_file()).then_some(resolved) + } + + fn settings_for(&self, source: &Path) -> Result { + let Some(name) = source.file_name().and_then(|n| n.to_str()) else { + return Ok(PageSettings::default()); + }; + let dir = source.parent().unwrap_or(&self.root); + Ok(self.dir_config(dir)?.settings_for(name)) + } + + fn dir_config(&self, dir: &Path) -> Result, Error> { + let path = dir.join(DIR_CONFIG); + if !path.is_file() { + return Ok(self.built_in.clone()); + } + self.dir_configs.get_or_insert_with(&path, || DirConfig::load(&path)) + } +} + +#[cfg(test)] +mod tests { + use std::fs; + + use super::*; + + /// Mirrors smolweb's `tests/conftest.py` fixture, so its assertions port + /// across directly. + fn fixture() -> (tempfile::TempDir, Site) { + let dir = tempfile::tempdir().unwrap(); + let root = dir.path(); + fs::write(root.join("index.md"), "# Home\n\nHello.\n").unwrap(); + fs::write(root.join("about.md"), "# About\n\nBody.\n").unwrap(); + fs::write(root.join("img.png"), b"\x89PNG\r\n\x1a\nfake").unwrap(); + fs::create_dir(root.join("dir")).unwrap(); + fs::write(root.join("dir/index.md"), "# Nested\n\nNested body.\n").unwrap(); + fs::write(root.join(".secret.md"), "# Hidden\n").unwrap(); + let site = Site::new(root).unwrap(); + (dir, site) + } + + #[track_caller] + fn document(site: &Site, path: &str) -> (String, PathBuf, PageSettings) { + match site.resolve(path).unwrap() { + Resolution::Found(Resource::Document { url, source, settings }) => { + (url, source, settings) + } + other => panic!("expected a document at {path}, got {other:?}"), + } + } + + /// Force a modification time change: filesystem granularity is coarse + /// enough that two writes in one test can share a timestamp. + fn bump_mtime(path: &Path) { + let later = std::time::SystemTime::now() + std::time::Duration::from_secs(5); + fs::File::options() + .write(true) + .open(path) + .unwrap() + .set_times(fs::FileTimes::new().set_accessed(later).set_modified(later)) + .unwrap(); + } + + #[track_caller] + fn assert_not_found(site: &Site, path: &str) { + match site.resolve(path).unwrap() { + Resolution::NotFound => {} + other => panic!("expected nothing at {path}, got {other:?}"), + } + } + + // -- Ported from TestPathTraversal ------------------------------------ + + #[test] + fn traversal_attempts_are_refused() { + let (_dir, site) = fixture(); + for path in [ + "/../../etc/passwd", + "/../../../../../../etc/passwd", + "/foo/../../etc/passwd", + // Percent-decoded to ".." before normalising, then clamped. + "/%2e%2e/etc/passwd", + ] { + assert_not_found(&site, path); + } + } + + #[test] + fn a_symlink_escaping_the_root_is_refused() { + let dir = tempfile::tempdir().unwrap(); + let outside = dir.path().join("outside"); + fs::create_dir(&outside).unwrap(); + fs::write(outside.join("secret.md"), "# Secret\n").unwrap(); + let root = dir.path().join("root"); + fs::create_dir(&root).unwrap(); + std::os::unix::fs::symlink(outside.join("secret.md"), root.join("escape.md")).unwrap(); + + let site = Site::new(&root).unwrap(); + assert_not_found(&site, "/escape"); + // The literal form is refused on the same grounds, not redirected. + assert_not_found(&site, "/escape.md"); + } + + #[test] + fn a_symlink_to_a_directory_outside_the_root_is_refused() { + let dir = tempfile::tempdir().unwrap(); + let outside = dir.path().join("outside"); + fs::create_dir(&outside).unwrap(); + fs::write(outside.join("index.md"), "# Secret\n").unwrap(); + let root = dir.path().join("root"); + fs::create_dir(&root).unwrap(); + std::os::unix::fs::symlink(&outside, root.join("link")).unwrap(); + + let site = Site::new(&root).unwrap(); + assert_not_found(&site, "/link"); + } + + #[test] + fn dotfile_paths_are_refused() { + let (_dir, site) = fixture(); + assert_not_found(&site, "/.secret"); + assert_not_found(&site, "/.secret.md"); + } + + #[test] + fn the_per_directory_config_is_never_served() { + let (dir, site) = fixture(); + fs::write(dir.path().join(DIR_CONFIG), "[defaults]\ncache_control = 60\n").unwrap(); + // It is not a dotfile by accident: without that rule it carries a dot in + // its name and so would resolve literally, like img.png does. + assert_not_found(&site, "/.itsybitsy.toml"); + } + + // -- Ported from TestResolutionRules ---------------------------------- + + #[test] + fn root_serves_index() { + let (dir, site) = fixture(); + let (url, source, _) = document(&site, "/"); + assert_eq!(url, "/"); + assert_eq!(source, dir.path().canonicalize().unwrap().join("index.md")); + } + + #[test] + fn an_extensionless_path_serves_the_md_file() { + let (_dir, site) = fixture(); + let (url, source, _) = document(&site, "/about"); + assert_eq!(url, "/about"); + assert_eq!(source.file_name().unwrap(), "about.md"); + } + + #[test] + fn a_directory_serves_its_index() { + let (_dir, site) = fixture(); + // Both forms resolve to the same page, and neither is redirected to the + // other. That is the behaviour being ported, not an oversight. + for path in ["/dir", "/dir/"] { + let (url, source, _) = document(&site, path); + assert_eq!(url, "/dir/"); + assert!(source.ends_with("dir/index.md"), "{source:?}"); + } + } + + #[test] + fn an_md_extension_redirects_to_the_canonical_url() { + let (_dir, site) = fixture(); + match site.resolve("/about.md").unwrap() { + Resolution::Redirect(location) => assert_eq!(location, "/about"), + other => panic!("expected a redirect, got {other:?}"), + } + match site.resolve("/dir/index.md").unwrap() { + Resolution::Redirect(location) => assert_eq!(location, "/dir/"), + other => panic!("expected a redirect, got {other:?}"), + } + } + + #[test] + fn a_missing_md_file_is_not_found() { + let (_dir, site) = fixture(); + assert_not_found(&site, "/nonexistent.md"); + } + + #[test] + fn a_raw_file_is_served_with_its_media_type() { + let (_dir, site) = fixture(); + match site.resolve("/img.png").unwrap() { + Resolution::Found(Resource::Raw { path, media_type }) => { + assert_eq!(media_type, "image/png"); + assert_eq!(path.file_name().unwrap(), "img.png"); + } + other => panic!("expected a raw file, got {other:?}"), + } + } + + #[test] + fn a_missing_path_is_not_found() { + let (_dir, site) = fixture(); + assert_not_found(&site, "/nope"); + } + + #[test] + fn a_bare_unresolvable_segment_resolves_to_nothing() { + // Regression carried over from smolweb: the Python reached this path + // through `rpartition("/")`, where a bare top-level segment yields an + // empty parent that must not be read as the root index. + let (_dir, site) = fixture(); + assert_not_found(&site, "/totally-unresolvable-segment"); + } + + #[test] + fn a_subpath_of_an_existing_document_is_not_found() { + // `/about/whatever` is not a sub-resource of about.md just because + // about.md exists. A format that invents sub-URLs claims them later. + let (_dir, site) = fixture(); + assert_not_found(&site, "/about/whatever"); + } + + #[test] + fn a_directory_without_an_index_is_not_found() { + let (dir, site) = fixture(); + fs::create_dir(dir.path().join("empty")).unwrap(); + assert_not_found(&site, "/empty"); + } + + // -- Ported from TestResolveFlat -------------------------------------- + + #[test] + fn a_canonical_redirect_is_resolved_not_bounced() { + let (_dir, site) = fixture(); + match site.resolve_flat("/about.md").unwrap() { + Resolution::Found(Resource::Document { url, .. }) => assert_eq!(url, "/about"), + other => panic!("expected the document itself, got {other:?}"), + } + } + + #[test] + fn resolve_flat_still_reports_a_missing_path() { + let (_dir, site) = fixture(); + match site.resolve_flat("/nope").unwrap() { + Resolution::NotFound => {} + other => panic!("expected nothing, got {other:?}"), + } + } + + #[test] + fn resolve_flat_passes_a_raw_file_straight_through() { + let (_dir, site) = fixture(); + match site.resolve_flat("/img.png").unwrap() { + Resolution::Found(Resource::Raw { media_type, .. }) => { + assert_eq!(media_type, "image/png"); + } + other => panic!("expected a raw file, got {other:?}"), + } + } + + // -- Per-directory settings ------------------------------------------- + + #[test] + fn settings_come_from_the_documents_own_directory() { + let (dir, site) = fixture(); + let root = dir.path(); + fs::write( + root.join(DIR_CONFIG), + "[defaults]\ncache_control = 3600\n\n[page.\"about.md\"]\ntitle = \"About Us\"\n", + ) + .unwrap(); + + let (_, _, about) = document(&site, "/about"); + assert_eq!(about.title.as_deref(), Some("About Us")); + assert_eq!(about.cache_control, Some(3600)); + + // index.md shares the directory defaults but has no title configured. + let (_, _, index) = document(&site, "/"); + assert_eq!(index.title, None); + assert_eq!(index.cache_control, Some(3600)); + + // The subdirectory does not inherit: it has no config of its own. + let (_, _, nested) = document(&site, "/dir"); + assert_eq!(nested, PageSettings::default()); + } + + #[test] + fn an_edited_directory_config_takes_effect() { + let (dir, site) = fixture(); + let path = dir.path().join(DIR_CONFIG); + fs::write(&path, "[defaults]\ncache_control = 60\n").unwrap(); + assert_eq!(document(&site, "/about").2.cache_control, Some(60)); + + fs::write(&path, "[defaults]\ncache_control = 120\n").unwrap(); + bump_mtime(&path); + + assert_eq!(document(&site, "/about").2.cache_control, Some(120)); + } + + #[test] + fn a_broken_directory_config_surfaces_as_an_error() { + let (dir, site) = fixture(); + fs::write(dir.path().join(DIR_CONFIG), "[defaults]\ncache_control = \"soon\"\n").unwrap(); + // Not a silent fall back to defaults: that is how a typo ships. + assert!(matches!(site.resolve("/about"), Err(Error::Toml { .. }))); + } + + #[test] + fn a_missing_root_is_rejected_at_construction() { + let dir = tempfile::tempdir().unwrap(); + assert!(matches!(Site::new(&dir.path().join("absent")), Err(Error::Io { .. }))); + } + + #[test] + fn a_file_as_a_root_is_rejected() { + let dir = tempfile::tempdir().unwrap(); + let file = dir.path().join("not-a-dir"); + fs::write(&file, "x").unwrap(); + assert!(matches!(Site::new(&file), Err(Error::Config { .. }))); + } +} diff --git a/core/src/siteset.rs b/core/src/siteset.rs new file mode 100644 index 0000000..9e61895 --- /dev/null +++ b/core/src/siteset.rs @@ -0,0 +1,150 @@ +//! Every site this server answers for, and how a request reaches one. +//! +//! Built once at startup from a validated configuration and immutable +//! thereafter, so a lookup is a hash probe with no locking. Sites whose roots +//! canonicalise to the same directory collapse into one [`Site`], which is what +//! lets two domains serving one folder share its caches. + +use std::collections::HashMap; +use std::sync::Arc; + +use crate::config::{ServerConfig, normalize_host}; +use crate::error::Error; +use crate::site::Site; + +pub struct SiteSet { + sites: Vec>, + /// Normalised host → index into `sites`. + by_host: HashMap, + /// Configured site name → index into `sites`. Two names can share one index. + by_name: HashMap, +} + +impl SiteSet { + /// Build from a configuration that [`ServerConfig::validate`] has accepted. + /// + /// Validation has already proved the roots canonicalise and that no host is + /// claimed twice, so the only work left is opening each site once. + pub fn build(config: &ServerConfig) -> Result { + let mut set = + SiteSet { sites: Vec::new(), by_host: HashMap::new(), by_name: HashMap::new() }; + + for (name, spec) in &config.site { + let site = Site::new(&spec.root)?; + // One `Site` per distinct root, so sites sharing a folder share its + // caches rather than each building their own. + let index = match set.sites.iter().position(|open| open.root() == site.root()) { + Some(index) => index, + None => { + set.sites.push(Arc::new(site)); + set.sites.len() - 1 + } + }; + set.by_name.insert(name.clone(), index); + + for raw in &spec.hosts { + let host = normalize_host(raw).ok_or_else(|| { + Error::config(format!("site '{name}': '{raw}' is not a usable hostname")) + })?; + set.by_host.insert(host, index); + } + } + + Ok(set) + } + + /// The site a request naming `host` is served from. `host` arrives as it was + /// on the wire; normalising it is this function's job. + pub fn lookup(&self, host: &str) -> Option<&Arc> { + let host = normalize_host(host)?; + self.sites.get(*self.by_host.get(&host)?) + } + + /// The site a listener named outright in its configuration. + pub fn by_name(&self, name: &str) -> Option<&Arc> { + self.sites.get(*self.by_name.get(name)?) + } + + /// How many distinct content roots are open, which is not the number of + /// configured sites when any of them share a root. + pub fn len(&self) -> usize { + self.sites.len() + } + + pub fn is_empty(&self) -> bool { + self.sites.is_empty() + } +} + +#[cfg(test)] +mod tests { + use std::fs; + use std::path::Path; + + use super::*; + + /// A config with two sites, whose roots are `first` and `second` under `dir`. + fn build(dir: &Path, first: &str, second: &str) -> SiteSet { + for name in [first, second] { + fs::create_dir_all(dir.join(name)).unwrap(); + } + let text = format!( + "version = 1\n\n\ + [site.one]\nroot = {:?}\nhosts = [\"one.test\", \"www.one.test\"]\n\n\ + [site.two]\nroot = {:?}\nhosts = [\"two.test\"]\n\n\ + [listener.web]\nprotocol = \"http\"\nbind = \":8080\"\nformats = [\"html\"]\n", + dir.join(first).to_str().unwrap(), + dir.join(second).to_str().unwrap(), + ); + let config: ServerConfig = toml::from_str(&text).unwrap(); + SiteSet::build(&config).unwrap() + } + + #[test] + fn routes_each_host_to_its_own_site() { + let dir = tempfile::tempdir().unwrap(); + let set = build(dir.path(), "one", "two"); + assert_eq!(set.len(), 2); + + let one = set.lookup("one.test").unwrap(); + assert!(one.root().ends_with("one")); + assert!(Arc::ptr_eq(one, set.lookup("www.one.test").unwrap())); + assert!(set.lookup("two.test").unwrap().root().ends_with("two")); + } + + #[test] + fn normalises_the_host_before_looking_it_up() { + let dir = tempfile::tempdir().unwrap(); + let set = build(dir.path(), "one", "two"); + for host in ["ONE.test", "one.test.", "one.test:8080", " one.test "] { + assert!(set.lookup(host).is_some(), "{host} should route"); + } + } + + #[test] + fn an_unknown_or_unusable_host_routes_nowhere() { + let dir = tempfile::tempdir().unwrap(); + let set = build(dir.path(), "one", "two"); + assert!(set.lookup("three.test").is_none()); + assert!(set.lookup("").is_none()); + // A control byte must not reach the map, let alone a log line. + assert!(set.lookup("one.test\r\nX: y").is_none()); + } + + #[test] + fn sites_sharing_a_root_share_one_site() { + let dir = tempfile::tempdir().unwrap(); + let set = build(dir.path(), "shared", "shared"); + assert_eq!(set.len(), 1, "one root, one Site, one set of caches"); + assert!(Arc::ptr_eq(set.lookup("one.test").unwrap(), set.lookup("two.test").unwrap())); + assert!(Arc::ptr_eq(set.by_name("one").unwrap(), set.by_name("two").unwrap())); + } + + #[test] + fn looks_a_site_up_by_its_configured_name() { + let dir = tempfile::tempdir().unwrap(); + let set = build(dir.path(), "one", "two"); + assert!(set.by_name("two").unwrap().root().ends_with("two")); + assert!(set.by_name("nope").is_none()); + } +}