docs: drop the predecessor comparisons from comments and tests
This commit is contained in:
parent
36faa4fb93
commit
19e7cfaca3
19 changed files with 65 additions and 73 deletions
|
|
@ -99,7 +99,6 @@ struct Request<'a> {
|
|||
/// 53 is "a resource at a domain not served by the server" and 59 is a request
|
||||
/// the server could not parse: a URL that parses but names another scheme or
|
||||
/// host is a proxy request, while one that does not parse is a bad request.
|
||||
/// smolweb answers 59 to both, including to an ordinary `https://` URL.
|
||||
fn parse(line: &str) -> Result<Request<'_>, Refusal> {
|
||||
let Some((scheme, rest)) = line.split_once("://") else {
|
||||
return Err(Refusal { status: 59, meta: "Bad request: an absolute URL is required" });
|
||||
|
|
@ -195,8 +194,8 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn another_scheme_is_a_proxy_request_not_a_parse_failure() {
|
||||
// smolweb answers 59 here, which tells a client its request was malformed
|
||||
// when it was merely for somewhere this server does not fetch from.
|
||||
// 59 would tell a client its request was malformed when it was merely for
|
||||
// somewhere this server does not fetch from.
|
||||
assert_eq!(refused("https://example.org/"), 53);
|
||||
assert_eq!(refused("gopher://example.org/"), 53);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ use itsybitsy_core::site::{Resolution, Resource};
|
|||
use crate::proto::for_log;
|
||||
use crate::serve::Listener;
|
||||
|
||||
/// Gopher selectors are short; the cap is smolweb's.
|
||||
/// Gopher selectors are short, and nothing in the protocol needs a long one.
|
||||
const MAX_REQUEST: usize = 512;
|
||||
|
||||
pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
|
||||
|
|
|
|||
|
|
@ -32,7 +32,8 @@ pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
|
|||
if !version.starts_with("HTTP/1.") {
|
||||
return bad_request(&mut stream);
|
||||
}
|
||||
// Origin-form only. smolweb's `urlsplit` would have mishandled the others.
|
||||
// Origin-form only; the other request-target forms are refused, not
|
||||
// half-supported.
|
||||
if !target.starts_with('/') {
|
||||
return bad_request(&mut stream);
|
||||
}
|
||||
|
|
@ -168,8 +169,8 @@ pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
|
|||
write_head(&mut stream, 200, "OK", media_type, meta.len(), &[])?;
|
||||
if !head_only {
|
||||
let mut file = std::fs::File::open(&path)?;
|
||||
// Streamed, not buffered: smolweb reads the whole file into
|
||||
// memory on every request.
|
||||
// Streamed, not buffered, so a large file costs the copy buffer
|
||||
// rather than its own size on every request.
|
||||
std::io::copy(&mut file, &mut stream)?;
|
||||
}
|
||||
Ok(())
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ use itsybitsy_core::site::{Resolution, Resource};
|
|||
use crate::proto::{for_log, read_line_capped};
|
||||
use crate::serve::Listener;
|
||||
|
||||
/// Nex requests are a single path; the cap is smolweb's.
|
||||
/// Nex requests are a single path, so the cap only has to be generous for one.
|
||||
const MAX_REQUEST: usize = 2048;
|
||||
|
||||
pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
//! Spartan: `HOST PATH LENGTH` in, a one-digit status and a body out.
|
||||
//!
|
||||
//! The host field is what smolweb discards; here it selects the virtual host,
|
||||
//! falling back to the listener's `default_site` when it names nothing known.
|
||||
//! The host field selects the virtual host, falling back to the listener's
|
||||
//! `default_site` when it names nothing known.
|
||||
|
||||
use std::io::{BufReader, Read, Write};
|
||||
use std::net::TcpStream;
|
||||
|
|
|
|||
|
|
@ -171,7 +171,7 @@ fn accept_loop(listener: Arc<Listener>, socket: TcpListener) {
|
|||
};
|
||||
|
||||
// At the cap, refuse immediately rather than queueing threads without
|
||||
// bound. smolweb has no cap at all.
|
||||
// bound.
|
||||
let open = listener.open.fetch_add(1, Ordering::SeqCst);
|
||||
if open >= listener.max_connections {
|
||||
listener.open.fetch_sub(1, Ordering::SeqCst);
|
||||
|
|
@ -194,8 +194,9 @@ fn handle(listener: &Listener, stream: TcpStream) {
|
|||
}
|
||||
|
||||
// One malformed document must not take the process down, so a panic in a
|
||||
// handler is caught and logged. This is the Rust equivalent of smolweb's
|
||||
// bare `except Exception`, except the cause is recorded rather than lost.
|
||||
// handler is caught and logged with its cause rather than lost. A stack
|
||||
// overflow is not a panic and aborts regardless, which is why the parser caps
|
||||
// how deeply a document may nest.
|
||||
let caught =
|
||||
std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| match listener.protocol {
|
||||
Protocol::Http => proto::http::serve(listener, stream),
|
||||
|
|
|
|||
|
|
@ -348,7 +348,7 @@ fn spartan_serves_gemtext() {
|
|||
}
|
||||
|
||||
#[test]
|
||||
fn spartan_routes_by_the_host_field_smolweb_discards() {
|
||||
fn spartan_routes_by_the_host_field() {
|
||||
let server = Server::start();
|
||||
assert!(server.send("spartan", b"one.test / 0\r\n").contains("# One"));
|
||||
assert!(server.send("spartan", b"two.test / 0\r\n").contains("# Two"));
|
||||
|
|
@ -560,8 +560,8 @@ fn nothing_outside_the_root_is_reachable_over_any_protocol() {
|
|||
|
||||
// -- Gopher --------------------------------------------------------------
|
||||
//
|
||||
// Ported from smolweb's tests/test_gopher.py, where the exact wire bytes are the
|
||||
// assertion: Gopher has no status line, so framing is all a client has.
|
||||
// The exact wire bytes are the assertion here: Gopher has no status line, so
|
||||
// framing is all a client has.
|
||||
|
||||
#[test]
|
||||
fn an_empty_selector_gets_a_one_item_menu() {
|
||||
|
|
@ -591,9 +591,8 @@ fn a_text_item_ends_with_the_lone_dot_terminator() {
|
|||
|
||||
#[test]
|
||||
fn a_missing_selector_is_a_well_formed_text_item() {
|
||||
// No status to report with, so the error is the item's content. smolweb says
|
||||
// "Not found." here and "Not found" over HTTP and Nex; one wording is used
|
||||
// across every protocol instead.
|
||||
// No status to report with, so the error is the item's content. One wording
|
||||
// is used across every protocol, so this matches the HTTP and Nex bodies.
|
||||
let server = Server::start();
|
||||
assert_eq!(server.send("gopher", b"/nope\r\n"), "Not found\n.\r\n");
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,8 +3,7 @@
|
|||
//! Invalidation is by modification time and length, which is what makes editing
|
||||
//! a file enough to see the change on the next request with no watcher and no
|
||||
//! restart. Values are built outside the lock, so two first hits on one file can
|
||||
//! both build it; the work is idempotent and the second insert wins, which is
|
||||
//! the trade the Python makes too.
|
||||
//! both build it; the work is idempotent and the second insert wins.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::fs;
|
||||
|
|
@ -109,9 +108,9 @@ mod tests {
|
|||
|
||||
use super::*;
|
||||
|
||||
/// Force a modification time change. Filesystem granularity is coarse
|
||||
/// enough that two writes in one test can otherwise share a timestamp,
|
||||
/// which is why the Python's live-reload test does the same thing.
|
||||
/// Force a modification time change. Filesystem granularity is coarse enough
|
||||
/// that two writes in one test can otherwise share a timestamp, which would
|
||||
/// make a stale value look correctly cached.
|
||||
fn bump_mtime(path: &Path) {
|
||||
let later = SystemTime::now() + Duration::from_secs(5);
|
||||
File::options()
|
||||
|
|
@ -137,7 +136,8 @@ mod tests {
|
|||
let first = cache.get_or_insert_with(&path, Stamp::of(&path).unwrap(), build).unwrap();
|
||||
let second = cache.get_or_insert_with(&path, Stamp::of(&path).unwrap(), build).unwrap();
|
||||
|
||||
// Identity, not just equality: the Python asserts `first is second`.
|
||||
// Identity, not equality: a repeat hit must return the cached value
|
||||
// rather than an equal rebuild.
|
||||
assert!(Arc::ptr_eq(&first, &second));
|
||||
assert_eq!(builds.load(Ordering::Relaxed), 1);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -670,8 +670,8 @@ impl Default for PageSettings {
|
|||
// the terminal edge.
|
||||
margin_left: 2,
|
||||
margin_right: 2,
|
||||
// One blank line between blocks. md2txt uses two, which reads as
|
||||
// double-spaced throughout.
|
||||
// One blank line between blocks; two reads as double-spaced
|
||||
// throughout.
|
||||
paragraph_spacing: 1,
|
||||
heading_styles: DEFAULT_HEADING_STYLES,
|
||||
blockquote_bars: true,
|
||||
|
|
|
|||
|
|
@ -1,10 +1,9 @@
|
|||
//! The one parsed representation every output format consumes.
|
||||
//!
|
||||
//! smolweb parses each document twice, with two hand-rolled regex parsers that
|
||||
//! share seven identical patterns but disagree on the edges: that is where the
|
||||
//! `{.card}` directive leaks into gemtext as literal text, and why the two
|
||||
//! libraries carry two divergent sets of defaults. Parsing once into this
|
||||
//! structure removes the class of bug rather than the instances.
|
||||
//! One parse, shared by every format. Parsing per format is how two outputs come
|
||||
//! to disagree about the same document: a directive one understands and another
|
||||
//! emits as literal text, or two sets of defaults that drift apart. Parsing once
|
||||
//! into this structure removes that class of bug rather than its instances.
|
||||
//!
|
||||
//! It is a flat block sequence rather than a tree of nodes because that is what
|
||||
//! the consumers want: WML packs a linear run of blocks into byte-budgeted
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
//! Media types for files served byte for byte.
|
||||
//!
|
||||
//! A fixed table rather than a system lookup: Python's `mimetypes` consults
|
||||
//! `/etc/mime.types` where it exists, so smolweb's `Content-Type` for the same
|
||||
//! file differs between hosts. Determinism is worth more here than coverage of
|
||||
//! the long tail, and an unknown extension has a correct answer anyway.
|
||||
//! A fixed table rather than a system lookup. A lookup reads `/etc/mime.types`
|
||||
//! where it exists, so the same file gets a different `Content-Type` depending on
|
||||
//! the host. Determinism is worth more here than coverage of the long tail, and
|
||||
//! an unknown extension has a correct answer anyway.
|
||||
|
||||
use std::path::Path;
|
||||
|
||||
|
|
|
|||
|
|
@ -427,8 +427,6 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn parses_setext_headings() {
|
||||
// wapdown's parser handles these and md2txt's does not, so unifying the
|
||||
// two parsers gains them for the text formats.
|
||||
assert_eq!(
|
||||
blocks("Title\n=====\n"),
|
||||
vec![Block::Heading { level: 1, inline: text("Title") }]
|
||||
|
|
|
|||
|
|
@ -82,8 +82,8 @@ fn encode_path(path: &str) -> String {
|
|||
/// Decode `%XX` escapes, leaving an invalid escape as the literal text it is.
|
||||
///
|
||||
/// Bytes that do not form valid UTF-8 become U+FFFD, which matches no filename,
|
||||
/// so a malformed target resolves to nothing rather than erroring. That matches
|
||||
/// Python's lossy `unquote` and is pinned by a test.
|
||||
/// so a malformed target resolves to nothing rather than erroring. A test pins
|
||||
/// that leniency, so it is not later tightened into an error.
|
||||
fn percent_decode(raw: &str) -> String {
|
||||
let bytes = raw.as_bytes();
|
||||
let mut out = Vec::with_capacity(bytes.len());
|
||||
|
|
@ -145,9 +145,8 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn clamps_traversal_at_the_root() {
|
||||
// Ported from smolweb's TestPathTraversal: these must never reach above
|
||||
// the root, and since nothing is mounted at the clamped path they
|
||||
// resolve to a path that simply does not exist.
|
||||
// These must never reach above the root, and since nothing is mounted at
|
||||
// the clamped path they resolve to a path that simply does not exist.
|
||||
assert_eq!(clean("/../../etc/passwd"), "etc/passwd");
|
||||
assert_eq!(clean("/../../../../../../etc/passwd"), "etc/passwd");
|
||||
assert_eq!(clean("/foo/../../etc/passwd"), "etc/passwd");
|
||||
|
|
@ -159,7 +158,7 @@ mod tests {
|
|||
// before the clamp runs, or it would be treated as a literal segment.
|
||||
assert_eq!(clean("/%2e%2e/etc/passwd"), "etc/passwd");
|
||||
assert_eq!(clean("/%2E%2E/etc/passwd"), "etc/passwd");
|
||||
// An encoded separator becomes a separator, as Python's unquote does.
|
||||
// An encoded separator becomes a real one, so normalising then sees it.
|
||||
assert_eq!(clean("/dir%2fpage"), "dir/page");
|
||||
assert_eq!(clean("/hello%20world"), "hello world");
|
||||
}
|
||||
|
|
|
|||
|
|
@ -7,10 +7,10 @@
|
|||
//! [`crate::directives`].
|
||||
//!
|
||||
//! Together with that module this is the only part of the pipeline that opens
|
||||
//! files, which gives the root-containment check exactly one home. That closes
|
||||
//! the traversal smolweb has: md2txt resolves an include target and checks only
|
||||
//! that it exists, so `{.include ../../../../etc/passwd}` in any served document
|
||||
//! reads and emits that file.
|
||||
//! files, which gives the root-containment check exactly one home. An include
|
||||
//! target is canonicalised and required to be inside the root, so a target of
|
||||
//! `../../../../etc/passwd` resolves to nothing instead of being read: checking
|
||||
//! only that a target exists is what makes includes a traversal.
|
||||
|
||||
use std::collections::BTreeSet;
|
||||
use std::fs;
|
||||
|
|
@ -24,7 +24,7 @@ use crate::error::{Error, IncludeReason};
|
|||
const MAX_DEPTH: usize = 16;
|
||||
/// Caps on the expanded result. The cycle set is per-*stack*, so a diamond —
|
||||
/// `a` includes `b` and `c`, both include `d` — fans out exponentially without
|
||||
/// ever repeating a file on one path. smolweb has nothing that stops this.
|
||||
/// ever repeating a file on one path, so only these caps stop it.
|
||||
const MAX_LINES: usize = 200_000;
|
||||
const MAX_BYTES: usize = 8 * 1024 * 1024;
|
||||
|
||||
|
|
@ -159,7 +159,7 @@ mod tests {
|
|||
assert_eq!(include_target("see ![[notes.md]] there"), None);
|
||||
assert_eq!(include_target("![[]]"), None);
|
||||
assert_eq!(include_target("![[unterminated"), None);
|
||||
// The directive smolweb also accepted is gone: one spelling, not two.
|
||||
// The brace-directive form is not accepted: one spelling, not two.
|
||||
assert_eq!(include_target("{.include notes.md}"), None);
|
||||
}
|
||||
|
||||
|
|
@ -263,8 +263,8 @@ mod tests {
|
|||
#[test]
|
||||
fn a_diamond_fan_out_is_stopped_by_the_size_cap() {
|
||||
// Each level doubles and no file repeats on any single path, so neither
|
||||
// the cycle set nor the depth cap catches it. smolweb expands this until
|
||||
// it runs out of memory.
|
||||
// the cycle set nor the depth cap catches it, which leaves the byte cap
|
||||
// as the only thing that stops it.
|
||||
let tree = Tree::new();
|
||||
tree.write("leaf.md", &"filler line\n".repeat(64));
|
||||
let mut previous = "leaf.md".to_string();
|
||||
|
|
|
|||
|
|
@ -3,8 +3,8 @@
|
|||
//! A format is a crate implementing [`Renderer`], registered at startup behind a
|
||||
//! cargo feature. The trait takes a parsed [`Doc`] rather than source text so
|
||||
//! that every format reads one parse: gemtext's `=>` link catalogue and the text
|
||||
//! formats' `[n]` references must agree about link identity and order, and in
|
||||
//! smolweb they can disagree because each library re-parses.
|
||||
//! formats' `[n]` references must agree about link identity and order, which
|
||||
//! cannot be relied on when each format parses the source for itself.
|
||||
//!
|
||||
//! A renderer must not open files or sockets. Includes and art are already
|
||||
//! resolved by the time it runs, which is what keeps the root-containment check
|
||||
|
|
|
|||
|
|
@ -267,8 +267,8 @@ mod tests {
|
|||
Site::new(root, registry(), vec!["stub".to_string()]).unwrap()
|
||||
}
|
||||
|
||||
/// Mirrors smolweb's `tests/conftest.py` fixture, so its assertions port
|
||||
/// across directly.
|
||||
/// One of each kind of thing a request can land on, shared by the resolution
|
||||
/// tests below.
|
||||
fn fixture() -> (tempfile::TempDir, Site) {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let root = dir.path();
|
||||
|
|
@ -449,9 +449,9 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn a_bare_unresolvable_segment_resolves_to_nothing() {
|
||||
// Regression carried over from smolweb: the Python reached this path
|
||||
// through `rpartition("/")`, where a bare top-level segment yields an
|
||||
// empty parent that must not be read as the root index.
|
||||
// The obvious implementation splits on the last `/`, where a bare
|
||||
// top-level segment yields an empty parent that must not then be read as
|
||||
// the root index.
|
||||
let (_dir, site) = fixture();
|
||||
assert_not_found(&site, "/totally-unresolvable-segment");
|
||||
}
|
||||
|
|
@ -568,9 +568,9 @@ mod tests {
|
|||
|
||||
#[cfg(test)]
|
||||
mod part_tests {
|
||||
//! Ported from smolweb's `TestWmlCardUrls`. A stub renderer stands in for a
|
||||
//! paginating format, so these rules are tested without the WML crate: core
|
||||
//! does not know which formats paginate, which is the point.
|
||||
//! A stub renderer stands in for a paginating format, so these rules are
|
||||
//! tested without the WML crate: core does not know which formats paginate,
|
||||
//! which is the point.
|
||||
|
||||
use std::fs;
|
||||
|
||||
|
|
|
|||
|
|
@ -2,9 +2,7 @@
|
|||
//!
|
||||
//! Plain text has no markup to carry emphasis, so it is dropped and the words
|
||||
//! kept. A link becomes `label (url)`: self-contained, and readable without
|
||||
//! scrolling to a reference list somewhere else. md2txt's `text` renderer emits
|
||||
//! numbered markers instead but never writes the list they point at, so the
|
||||
//! numbers lead nowhere.
|
||||
//! scrolling to a reference list somewhere else.
|
||||
|
||||
use itsybitsy_core::ir::{Doc, Inline};
|
||||
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
//! Fixed-width plain text, for Nex and later Gopher.
|
||||
//!
|
||||
//! One renderer, not two. md2txt ships a `text` and a `nex` renderer that differ
|
||||
//! in exactly two things — whether headings get FIGlet banners, and whether links
|
||||
//! are inlined or numbered — and both of those are now configuration. Its
|
||||
//! numbered form never writes the reference list its numbers point at, so the
|
||||
//! inline form is the only one that works and is the default here.
|
||||
//! One renderer, not two: whether a heading gets a FIGlet banner and how a link
|
||||
//! is written are both configuration, so Nex and Gopher are this renderer with
|
||||
//! different settings rather than renderers of their own. A link is inlined as
|
||||
//! `label (url)` rather than numbered, because a numbered marker is only useful
|
||||
//! with a reference list to point at.
|
||||
//!
|
||||
//! Unlike gemtext this wraps, because Nex and Gopher clients do not.
|
||||
|
||||
|
|
@ -115,7 +115,7 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn one_blank_line_separates_blocks_by_default() {
|
||||
// md2txt emits two, which reads as double-spaced throughout.
|
||||
// Two would read as double-spaced throughout.
|
||||
assert_eq!(render("a\n\nb\n"), "a\n\nb\n");
|
||||
}
|
||||
|
||||
|
|
@ -184,7 +184,6 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn tables_are_rendered_as_aligned_columns() {
|
||||
// md2txt drops tables entirely; this is the flaw that fixes.
|
||||
assert_eq!(
|
||||
render("| Format | Port |\n| --- | --- |\n| Nex | 1900 |\n"),
|
||||
"Format Port\n------ ----\nNex 1900\n"
|
||||
|
|
|
|||
|
|
@ -184,7 +184,6 @@ mod tests {
|
|||
|
||||
#[test]
|
||||
fn tables_are_rendered_because_wml_has_them() {
|
||||
// wapdown's own parser produces no tables, so this is new output.
|
||||
assert_eq!(
|
||||
render("| a | b |\n| --- | --- |\n| 1 | 2 |\n"),
|
||||
"<table columns=\"2\">\n<tr><td>a</td><td>b</td></tr>\n<tr><td>1</td><td>2</td></tr>\n</table>\n"
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue