docs: drop the predecessor comparisons from comments and tests

This commit is contained in:
randogoth 2026-10-06 10:27:46 +03:00
parent 36faa4fb93
commit 19e7cfaca3
19 changed files with 65 additions and 73 deletions

View file

@ -99,7 +99,6 @@ struct Request<'a> {
/// 53 is "a resource at a domain not served by the server" and 59 is a request /// 53 is "a resource at a domain not served by the server" and 59 is a request
/// the server could not parse: a URL that parses but names another scheme or /// the server could not parse: a URL that parses but names another scheme or
/// host is a proxy request, while one that does not parse is a bad request. /// host is a proxy request, while one that does not parse is a bad request.
/// smolweb answers 59 to both, including to an ordinary `https://` URL.
fn parse(line: &str) -> Result<Request<'_>, Refusal> { fn parse(line: &str) -> Result<Request<'_>, Refusal> {
let Some((scheme, rest)) = line.split_once("://") else { let Some((scheme, rest)) = line.split_once("://") else {
return Err(Refusal { status: 59, meta: "Bad request: an absolute URL is required" }); return Err(Refusal { status: 59, meta: "Bad request: an absolute URL is required" });
@ -195,8 +194,8 @@ mod tests {
#[test] #[test]
fn another_scheme_is_a_proxy_request_not_a_parse_failure() { fn another_scheme_is_a_proxy_request_not_a_parse_failure() {
// smolweb answers 59 here, which tells a client its request was malformed // 59 would tell a client its request was malformed when it was merely for
// when it was merely for somewhere this server does not fetch from. // somewhere this server does not fetch from.
assert_eq!(refused("https://example.org/"), 53); assert_eq!(refused("https://example.org/"), 53);
assert_eq!(refused("gopher://example.org/"), 53); assert_eq!(refused("gopher://example.org/"), 53);
} }

View file

@ -16,7 +16,7 @@ use itsybitsy_core::site::{Resolution, Resource};
use crate::proto::for_log; use crate::proto::for_log;
use crate::serve::Listener; use crate::serve::Listener;
/// Gopher selectors are short; the cap is smolweb's. /// Gopher selectors are short, and nothing in the protocol needs a long one.
const MAX_REQUEST: usize = 512; const MAX_REQUEST: usize = 512;
pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> { pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {

View file

@ -32,7 +32,8 @@ pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
if !version.starts_with("HTTP/1.") { if !version.starts_with("HTTP/1.") {
return bad_request(&mut stream); return bad_request(&mut stream);
} }
// Origin-form only. smolweb's `urlsplit` would have mishandled the others. // Origin-form only; the other request-target forms are refused, not
// half-supported.
if !target.starts_with('/') { if !target.starts_with('/') {
return bad_request(&mut stream); return bad_request(&mut stream);
} }
@ -168,8 +169,8 @@ pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
write_head(&mut stream, 200, "OK", media_type, meta.len(), &[])?; write_head(&mut stream, 200, "OK", media_type, meta.len(), &[])?;
if !head_only { if !head_only {
let mut file = std::fs::File::open(&path)?; let mut file = std::fs::File::open(&path)?;
// Streamed, not buffered: smolweb reads the whole file into // Streamed, not buffered, so a large file costs the copy buffer
// memory on every request. // rather than its own size on every request.
std::io::copy(&mut file, &mut stream)?; std::io::copy(&mut file, &mut stream)?;
} }
Ok(()) Ok(())

View file

@ -13,7 +13,7 @@ use itsybitsy_core::site::{Resolution, Resource};
use crate::proto::{for_log, read_line_capped}; use crate::proto::{for_log, read_line_capped};
use crate::serve::Listener; use crate::serve::Listener;
/// Nex requests are a single path; the cap is smolweb's. /// Nex requests are a single path, so the cap only has to be generous for one.
const MAX_REQUEST: usize = 2048; const MAX_REQUEST: usize = 2048;
pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> { pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {

View file

@ -1,7 +1,7 @@
//! Spartan: `HOST PATH LENGTH` in, a one-digit status and a body out. //! Spartan: `HOST PATH LENGTH` in, a one-digit status and a body out.
//! //!
//! The host field is what smolweb discards; here it selects the virtual host, //! The host field selects the virtual host, falling back to the listener's
//! falling back to the listener's `default_site` when it names nothing known. //! `default_site` when it names nothing known.
use std::io::{BufReader, Read, Write}; use std::io::{BufReader, Read, Write};
use std::net::TcpStream; use std::net::TcpStream;

View file

@ -171,7 +171,7 @@ fn accept_loop(listener: Arc<Listener>, socket: TcpListener) {
}; };
// At the cap, refuse immediately rather than queueing threads without // At the cap, refuse immediately rather than queueing threads without
// bound. smolweb has no cap at all. // bound.
let open = listener.open.fetch_add(1, Ordering::SeqCst); let open = listener.open.fetch_add(1, Ordering::SeqCst);
if open >= listener.max_connections { if open >= listener.max_connections {
listener.open.fetch_sub(1, Ordering::SeqCst); listener.open.fetch_sub(1, Ordering::SeqCst);
@ -194,8 +194,9 @@ fn handle(listener: &Listener, stream: TcpStream) {
} }
// One malformed document must not take the process down, so a panic in a // One malformed document must not take the process down, so a panic in a
// handler is caught and logged. This is the Rust equivalent of smolweb's // handler is caught and logged with its cause rather than lost. A stack
// bare `except Exception`, except the cause is recorded rather than lost. // overflow is not a panic and aborts regardless, which is why the parser caps
// how deeply a document may nest.
let caught = let caught =
std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| match listener.protocol { std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| match listener.protocol {
Protocol::Http => proto::http::serve(listener, stream), Protocol::Http => proto::http::serve(listener, stream),

View file

@ -348,7 +348,7 @@ fn spartan_serves_gemtext() {
} }
#[test] #[test]
fn spartan_routes_by_the_host_field_smolweb_discards() { fn spartan_routes_by_the_host_field() {
let server = Server::start(); let server = Server::start();
assert!(server.send("spartan", b"one.test / 0\r\n").contains("# One")); assert!(server.send("spartan", b"one.test / 0\r\n").contains("# One"));
assert!(server.send("spartan", b"two.test / 0\r\n").contains("# Two")); assert!(server.send("spartan", b"two.test / 0\r\n").contains("# Two"));
@ -560,8 +560,8 @@ fn nothing_outside_the_root_is_reachable_over_any_protocol() {
// -- Gopher -------------------------------------------------------------- // -- Gopher --------------------------------------------------------------
// //
// Ported from smolweb's tests/test_gopher.py, where the exact wire bytes are the // The exact wire bytes are the assertion here: Gopher has no status line, so
// assertion: Gopher has no status line, so framing is all a client has. // framing is all a client has.
#[test] #[test]
fn an_empty_selector_gets_a_one_item_menu() { fn an_empty_selector_gets_a_one_item_menu() {
@ -591,9 +591,8 @@ fn a_text_item_ends_with_the_lone_dot_terminator() {
#[test] #[test]
fn a_missing_selector_is_a_well_formed_text_item() { fn a_missing_selector_is_a_well_formed_text_item() {
// No status to report with, so the error is the item's content. smolweb says // No status to report with, so the error is the item's content. One wording
// "Not found." here and "Not found" over HTTP and Nex; one wording is used // is used across every protocol, so this matches the HTTP and Nex bodies.
// across every protocol instead.
let server = Server::start(); let server = Server::start();
assert_eq!(server.send("gopher", b"/nope\r\n"), "Not found\n.\r\n"); assert_eq!(server.send("gopher", b"/nope\r\n"), "Not found\n.\r\n");
} }

View file

@ -3,8 +3,7 @@
//! Invalidation is by modification time and length, which is what makes editing //! Invalidation is by modification time and length, which is what makes editing
//! a file enough to see the change on the next request with no watcher and no //! a file enough to see the change on the next request with no watcher and no
//! restart. Values are built outside the lock, so two first hits on one file can //! restart. Values are built outside the lock, so two first hits on one file can
//! both build it; the work is idempotent and the second insert wins, which is //! both build it; the work is idempotent and the second insert wins.
//! the trade the Python makes too.
use std::collections::HashMap; use std::collections::HashMap;
use std::fs; use std::fs;
@ -109,9 +108,9 @@ mod tests {
use super::*; use super::*;
/// Force a modification time change. Filesystem granularity is coarse /// Force a modification time change. Filesystem granularity is coarse enough
/// enough that two writes in one test can otherwise share a timestamp, /// that two writes in one test can otherwise share a timestamp, which would
/// which is why the Python's live-reload test does the same thing. /// make a stale value look correctly cached.
fn bump_mtime(path: &Path) { fn bump_mtime(path: &Path) {
let later = SystemTime::now() + Duration::from_secs(5); let later = SystemTime::now() + Duration::from_secs(5);
File::options() File::options()
@ -137,7 +136,8 @@ mod tests {
let first = cache.get_or_insert_with(&path, Stamp::of(&path).unwrap(), build).unwrap(); let first = cache.get_or_insert_with(&path, Stamp::of(&path).unwrap(), build).unwrap();
let second = cache.get_or_insert_with(&path, Stamp::of(&path).unwrap(), build).unwrap(); let second = cache.get_or_insert_with(&path, Stamp::of(&path).unwrap(), build).unwrap();
// Identity, not just equality: the Python asserts `first is second`. // Identity, not equality: a repeat hit must return the cached value
// rather than an equal rebuild.
assert!(Arc::ptr_eq(&first, &second)); assert!(Arc::ptr_eq(&first, &second));
assert_eq!(builds.load(Ordering::Relaxed), 1); assert_eq!(builds.load(Ordering::Relaxed), 1);
} }

View file

@ -670,8 +670,8 @@ impl Default for PageSettings {
// the terminal edge. // the terminal edge.
margin_left: 2, margin_left: 2,
margin_right: 2, margin_right: 2,
// One blank line between blocks. md2txt uses two, which reads as // One blank line between blocks; two reads as double-spaced
// double-spaced throughout. // throughout.
paragraph_spacing: 1, paragraph_spacing: 1,
heading_styles: DEFAULT_HEADING_STYLES, heading_styles: DEFAULT_HEADING_STYLES,
blockquote_bars: true, blockquote_bars: true,

View file

@ -1,10 +1,9 @@
//! The one parsed representation every output format consumes. //! The one parsed representation every output format consumes.
//! //!
//! smolweb parses each document twice, with two hand-rolled regex parsers that //! One parse, shared by every format. Parsing per format is how two outputs come
//! share seven identical patterns but disagree on the edges: that is where the //! to disagree about the same document: a directive one understands and another
//! `{.card}` directive leaks into gemtext as literal text, and why the two //! emits as literal text, or two sets of defaults that drift apart. Parsing once
//! libraries carry two divergent sets of defaults. Parsing once into this //! into this structure removes that class of bug rather than its instances.
//! structure removes the class of bug rather than the instances.
//! //!
//! It is a flat block sequence rather than a tree of nodes because that is what //! It is a flat block sequence rather than a tree of nodes because that is what
//! the consumers want: WML packs a linear run of blocks into byte-budgeted //! the consumers want: WML packs a linear run of blocks into byte-budgeted

View file

@ -1,9 +1,9 @@
//! Media types for files served byte for byte. //! Media types for files served byte for byte.
//! //!
//! A fixed table rather than a system lookup: Python's `mimetypes` consults //! A fixed table rather than a system lookup. A lookup reads `/etc/mime.types`
//! `/etc/mime.types` where it exists, so smolweb's `Content-Type` for the same //! where it exists, so the same file gets a different `Content-Type` depending on
//! file differs between hosts. Determinism is worth more here than coverage of //! the host. Determinism is worth more here than coverage of the long tail, and
//! the long tail, and an unknown extension has a correct answer anyway. //! an unknown extension has a correct answer anyway.
use std::path::Path; use std::path::Path;

View file

@ -427,8 +427,6 @@ mod tests {
#[test] #[test]
fn parses_setext_headings() { fn parses_setext_headings() {
// wapdown's parser handles these and md2txt's does not, so unifying the
// two parsers gains them for the text formats.
assert_eq!( assert_eq!(
blocks("Title\n=====\n"), blocks("Title\n=====\n"),
vec![Block::Heading { level: 1, inline: text("Title") }] vec![Block::Heading { level: 1, inline: text("Title") }]

View file

@ -82,8 +82,8 @@ fn encode_path(path: &str) -> String {
/// Decode `%XX` escapes, leaving an invalid escape as the literal text it is. /// Decode `%XX` escapes, leaving an invalid escape as the literal text it is.
/// ///
/// Bytes that do not form valid UTF-8 become U+FFFD, which matches no filename, /// Bytes that do not form valid UTF-8 become U+FFFD, which matches no filename,
/// so a malformed target resolves to nothing rather than erroring. That matches /// so a malformed target resolves to nothing rather than erroring. A test pins
/// Python's lossy `unquote` and is pinned by a test. /// that leniency, so it is not later tightened into an error.
fn percent_decode(raw: &str) -> String { fn percent_decode(raw: &str) -> String {
let bytes = raw.as_bytes(); let bytes = raw.as_bytes();
let mut out = Vec::with_capacity(bytes.len()); let mut out = Vec::with_capacity(bytes.len());
@ -145,9 +145,8 @@ mod tests {
#[test] #[test]
fn clamps_traversal_at_the_root() { fn clamps_traversal_at_the_root() {
// Ported from smolweb's TestPathTraversal: these must never reach above // These must never reach above the root, and since nothing is mounted at
// the root, and since nothing is mounted at the clamped path they // the clamped path they resolve to a path that simply does not exist.
// resolve to a path that simply does not exist.
assert_eq!(clean("/../../etc/passwd"), "etc/passwd"); assert_eq!(clean("/../../etc/passwd"), "etc/passwd");
assert_eq!(clean("/../../../../../../etc/passwd"), "etc/passwd"); assert_eq!(clean("/../../../../../../etc/passwd"), "etc/passwd");
assert_eq!(clean("/foo/../../etc/passwd"), "etc/passwd"); assert_eq!(clean("/foo/../../etc/passwd"), "etc/passwd");
@ -159,7 +158,7 @@ mod tests {
// before the clamp runs, or it would be treated as a literal segment. // before the clamp runs, or it would be treated as a literal segment.
assert_eq!(clean("/%2e%2e/etc/passwd"), "etc/passwd"); assert_eq!(clean("/%2e%2e/etc/passwd"), "etc/passwd");
assert_eq!(clean("/%2E%2E/etc/passwd"), "etc/passwd"); assert_eq!(clean("/%2E%2E/etc/passwd"), "etc/passwd");
// An encoded separator becomes a separator, as Python's unquote does. // An encoded separator becomes a real one, so normalising then sees it.
assert_eq!(clean("/dir%2fpage"), "dir/page"); assert_eq!(clean("/dir%2fpage"), "dir/page");
assert_eq!(clean("/hello%20world"), "hello world"); assert_eq!(clean("/hello%20world"), "hello world");
} }

View file

@ -7,10 +7,10 @@
//! [`crate::directives`]. //! [`crate::directives`].
//! //!
//! Together with that module this is the only part of the pipeline that opens //! Together with that module this is the only part of the pipeline that opens
//! files, which gives the root-containment check exactly one home. That closes //! files, which gives the root-containment check exactly one home. An include
//! the traversal smolweb has: md2txt resolves an include target and checks only //! target is canonicalised and required to be inside the root, so a target of
//! that it exists, so `{.include ../../../../etc/passwd}` in any served document //! `../../../../etc/passwd` resolves to nothing instead of being read: checking
//! reads and emits that file. //! only that a target exists is what makes includes a traversal.
use std::collections::BTreeSet; use std::collections::BTreeSet;
use std::fs; use std::fs;
@ -24,7 +24,7 @@ use crate::error::{Error, IncludeReason};
const MAX_DEPTH: usize = 16; const MAX_DEPTH: usize = 16;
/// Caps on the expanded result. The cycle set is per-*stack*, so a diamond — /// Caps on the expanded result. The cycle set is per-*stack*, so a diamond —
/// `a` includes `b` and `c`, both include `d` — fans out exponentially without /// `a` includes `b` and `c`, both include `d` — fans out exponentially without
/// ever repeating a file on one path. smolweb has nothing that stops this. /// ever repeating a file on one path, so only these caps stop it.
const MAX_LINES: usize = 200_000; const MAX_LINES: usize = 200_000;
const MAX_BYTES: usize = 8 * 1024 * 1024; const MAX_BYTES: usize = 8 * 1024 * 1024;
@ -159,7 +159,7 @@ mod tests {
assert_eq!(include_target("see ![[notes.md]] there"), None); assert_eq!(include_target("see ![[notes.md]] there"), None);
assert_eq!(include_target("![[]]"), None); assert_eq!(include_target("![[]]"), None);
assert_eq!(include_target("![[unterminated"), None); assert_eq!(include_target("![[unterminated"), None);
// The directive smolweb also accepted is gone: one spelling, not two. // The brace-directive form is not accepted: one spelling, not two.
assert_eq!(include_target("{.include notes.md}"), None); assert_eq!(include_target("{.include notes.md}"), None);
} }
@ -263,8 +263,8 @@ mod tests {
#[test] #[test]
fn a_diamond_fan_out_is_stopped_by_the_size_cap() { fn a_diamond_fan_out_is_stopped_by_the_size_cap() {
// Each level doubles and no file repeats on any single path, so neither // Each level doubles and no file repeats on any single path, so neither
// the cycle set nor the depth cap catches it. smolweb expands this until // the cycle set nor the depth cap catches it, which leaves the byte cap
// it runs out of memory. // as the only thing that stops it.
let tree = Tree::new(); let tree = Tree::new();
tree.write("leaf.md", &"filler line\n".repeat(64)); tree.write("leaf.md", &"filler line\n".repeat(64));
let mut previous = "leaf.md".to_string(); let mut previous = "leaf.md".to_string();

View file

@ -3,8 +3,8 @@
//! A format is a crate implementing [`Renderer`], registered at startup behind a //! A format is a crate implementing [`Renderer`], registered at startup behind a
//! cargo feature. The trait takes a parsed [`Doc`] rather than source text so //! cargo feature. The trait takes a parsed [`Doc`] rather than source text so
//! that every format reads one parse: gemtext's `=>` link catalogue and the text //! that every format reads one parse: gemtext's `=>` link catalogue and the text
//! formats' `[n]` references must agree about link identity and order, and in //! formats' `[n]` references must agree about link identity and order, which
//! smolweb they can disagree because each library re-parses. //! cannot be relied on when each format parses the source for itself.
//! //!
//! A renderer must not open files or sockets. Includes and art are already //! A renderer must not open files or sockets. Includes and art are already
//! resolved by the time it runs, which is what keeps the root-containment check //! resolved by the time it runs, which is what keeps the root-containment check

View file

@ -267,8 +267,8 @@ mod tests {
Site::new(root, registry(), vec!["stub".to_string()]).unwrap() Site::new(root, registry(), vec!["stub".to_string()]).unwrap()
} }
/// Mirrors smolweb's `tests/conftest.py` fixture, so its assertions port /// One of each kind of thing a request can land on, shared by the resolution
/// across directly. /// tests below.
fn fixture() -> (tempfile::TempDir, Site) { fn fixture() -> (tempfile::TempDir, Site) {
let dir = tempfile::tempdir().unwrap(); let dir = tempfile::tempdir().unwrap();
let root = dir.path(); let root = dir.path();
@ -449,9 +449,9 @@ mod tests {
#[test] #[test]
fn a_bare_unresolvable_segment_resolves_to_nothing() { fn a_bare_unresolvable_segment_resolves_to_nothing() {
// Regression carried over from smolweb: the Python reached this path // The obvious implementation splits on the last `/`, where a bare
// through `rpartition("/")`, where a bare top-level segment yields an // top-level segment yields an empty parent that must not then be read as
// empty parent that must not be read as the root index. // the root index.
let (_dir, site) = fixture(); let (_dir, site) = fixture();
assert_not_found(&site, "/totally-unresolvable-segment"); assert_not_found(&site, "/totally-unresolvable-segment");
} }
@ -568,9 +568,9 @@ mod tests {
#[cfg(test)] #[cfg(test)]
mod part_tests { mod part_tests {
//! Ported from smolweb's `TestWmlCardUrls`. A stub renderer stands in for a //! A stub renderer stands in for a paginating format, so these rules are
//! paginating format, so these rules are tested without the WML crate: core //! tested without the WML crate: core does not know which formats paginate,
//! does not know which formats paginate, which is the point. //! which is the point.
use std::fs; use std::fs;

View file

@ -2,9 +2,7 @@
//! //!
//! Plain text has no markup to carry emphasis, so it is dropped and the words //! Plain text has no markup to carry emphasis, so it is dropped and the words
//! kept. A link becomes `label (url)`: self-contained, and readable without //! kept. A link becomes `label (url)`: self-contained, and readable without
//! scrolling to a reference list somewhere else. md2txt's `text` renderer emits //! scrolling to a reference list somewhere else.
//! numbered markers instead but never writes the list they point at, so the
//! numbers lead nowhere.
use itsybitsy_core::ir::{Doc, Inline}; use itsybitsy_core::ir::{Doc, Inline};

View file

@ -1,10 +1,10 @@
//! Fixed-width plain text, for Nex and later Gopher. //! Fixed-width plain text, for Nex and later Gopher.
//! //!
//! One renderer, not two. md2txt ships a `text` and a `nex` renderer that differ //! One renderer, not two: whether a heading gets a FIGlet banner and how a link
//! in exactly two things — whether headings get FIGlet banners, and whether links //! is written are both configuration, so Nex and Gopher are this renderer with
//! are inlined or numbered — and both of those are now configuration. Its //! different settings rather than renderers of their own. A link is inlined as
//! numbered form never writes the reference list its numbers point at, so the //! `label (url)` rather than numbered, because a numbered marker is only useful
//! inline form is the only one that works and is the default here. //! with a reference list to point at.
//! //!
//! Unlike gemtext this wraps, because Nex and Gopher clients do not. //! Unlike gemtext this wraps, because Nex and Gopher clients do not.
@ -115,7 +115,7 @@ mod tests {
#[test] #[test]
fn one_blank_line_separates_blocks_by_default() { fn one_blank_line_separates_blocks_by_default() {
// md2txt emits two, which reads as double-spaced throughout. // Two would read as double-spaced throughout.
assert_eq!(render("a\n\nb\n"), "a\n\nb\n"); assert_eq!(render("a\n\nb\n"), "a\n\nb\n");
} }
@ -184,7 +184,6 @@ mod tests {
#[test] #[test]
fn tables_are_rendered_as_aligned_columns() { fn tables_are_rendered_as_aligned_columns() {
// md2txt drops tables entirely; this is the flaw that fixes.
assert_eq!( assert_eq!(
render("| Format | Port |\n| --- | --- |\n| Nex | 1900 |\n"), render("| Format | Port |\n| --- | --- |\n| Nex | 1900 |\n"),
"Format Port\n------ ----\nNex 1900\n" "Format Port\n------ ----\nNex 1900\n"

View file

@ -184,7 +184,6 @@ mod tests {
#[test] #[test]
fn tables_are_rendered_because_wml_has_them() { fn tables_are_rendered_because_wml_has_them() {
// wapdown's own parser produces no tables, so this is new output.
assert_eq!( assert_eq!(
render("| a | b |\n| --- | --- |\n| 1 | 2 |\n"), render("| a | b |\n| --- | --- |\n| 1 | 2 |\n"),
"<table columns=\"2\">\n<tr><td>a</td><td>b</td></tr>\n<tr><td>1</td><td>2</td></tr>\n</table>\n" "<table columns=\"2\">\n<tr><td>a</td><td>b</td></tr>\n<tr><td>1</td><td>2</td></tr>\n</table>\n"