itsybitsy/bin/src/proto/gemini.rs

224 lines
8.9 KiB
Rust
Raw Normal View History

//! Gemini: one absolute URL in, a status line and a body out.
//!
//! TLS is terminated in front of this listener, by `stunnel`, `ghostunnel` or any
//! other TLS wrapper, so what arrives here is plaintext. That keeps a TLS stack,
//! certificate loading and SNI out of the binary entirely; the protocol below is
//! the whole of Gemini that is not TLS. Two consequences are worth knowing: every
//! connection appears to come from the terminator unless it speaks the PROXY
//! protocol, and a client certificate can never reach us, so statuses 60 to 62
//! are not implementable here. itsybitsy serves static documents and has nothing
//! to authenticate, so only the logs are poorer for it.
use std::io::{BufReader, Write};
use std::net::TcpStream;
use anyhow::Result;
use itsybitsy_core::site::{Resolution, Resource};
use crate::proto::{for_log, read_line_capped};
use crate::serve::Listener;
/// The spec caps a request URL at 1024 bytes; this cap counts its CRLF too.
const MAX_REQUEST: usize = 1026;
pub fn serve(listener: &Listener, mut stream: TcpStream) -> Result<()> {
let line = {
let mut reader = BufReader::new(stream.try_clone()?);
read_line_capped(&mut reader, MAX_REQUEST)
};
let Some(line) = line else {
log::warn!("{}: unreadable or oversized request", listener.name);
return header(&mut stream, 59, "Bad request");
};
let request = match parse(&line) {
Ok(request) => request,
Err(Refusal { status, meta }) => {
log::info!("{} gemini refused {}: {meta}", listener.name, for_log(&line));
return header(&mut stream, status, meta);
}
};
// A host this server does not answer for is a request to fetch from
// elsewhere, which is what 53 is for; falling back to `default_site` first
// matches how the other host-addressed listeners behave.
let Some(site) = listener.site_for(request.host) else {
log::info!("{} gemini unknown host {}", listener.name, for_log(request.host));
return header(&mut stream, 53, "Proxy request refused");
};
log::info!("{} gemini {} {}", listener.name, for_log(request.host), for_log(request.path));
let format = &listener.formats[0];
match site.resolve(request.path) {
Ok(Resolution::Found(Resource::Document { page, .. })) => match page.body(format) {
Some(body) => body_response(&mut stream, listener.media_type(format), body),
None => header(&mut stream, 40, "Temporary failure"),
},
Ok(Resolution::Found(Resource::Part { parent_url, slug, page })) => {
match page.part(format, &slug) {
Some(body) => body_response(&mut stream, listener.media_type(format), body),
// Sub-documents belong to the paginating formats; Gemini serves
// the whole document instead of a card that does not exist here.
None => header(&mut stream, 31, &parent_url),
}
}
Ok(Resolution::Found(Resource::Raw { path, media_type })) => {
write!(stream, "20 {media_type}\r\n")?;
let mut file = std::fs::File::open(&path)?;
std::io::copy(&mut file, &mut stream)?;
Ok(())
}
Ok(Resolution::Redirect(location)) => header(&mut stream, 31, &location),
Ok(Resolution::NotFound) => header(&mut stream, 51, "Not found"),
Err(err) => {
log::warn!("{} gemini {}: {err}", listener.name, for_log(request.path));
// 40 rather than 50: the cause is a document or a directory config an
// operator can fix, so a client is right to try again later.
header(&mut stream, 40, "Temporary failure")
}
}
}
/// A request that will not be served, and the status saying why.
struct Refusal {
status: u8,
meta: &'static str,
}
struct Request<'a> {
/// The URL's authority, port included. Host normalisation strips the port.
host: &'a str,
path: &'a str,
}
/// Pick apart `gemini://host/path?query`.
///
/// Hand-rolled rather than taking a URL crate as a dependency: only the scheme,
/// the authority and the remainder are wanted, and path cleaning already drops
/// the query. The split between 53 and 59 follows the spec's own wording, where
/// 53 is "a resource at a domain not served by the server" and 59 is a request
/// the server could not parse: a URL that parses but names another scheme or
/// host is a proxy request, while one that does not parse is a bad request.
/// smolweb answers 59 to both, including to an ordinary `https://` URL.
fn parse(line: &str) -> Result<Request<'_>, Refusal> {
let Some((scheme, rest)) = line.split_once("://") else {
return Err(Refusal { status: 59, meta: "Bad request: an absolute URL is required" });
};
if !scheme.eq_ignore_ascii_case("gemini") {
return Err(Refusal { status: 53, meta: "Proxy request refused" });
}
let end = rest.find(['/', '?', '#']).unwrap_or(rest.len());
let (host, path) = rest.split_at(end);
if host.is_empty() {
return Err(Refusal { status: 59, meta: "Bad request: the URL names no host" });
}
// The spec forbids userinfo outright and forbids a client from sending a
// fragment, so neither is merely a request for somewhere else: 53 would
// suggest the URL was fine and only the host was foreign.
if host.contains('@') {
return Err(Refusal { status: 59, meta: "Bad request: userinfo is not allowed" });
}
if path.contains('#') {
return Err(Refusal { status: 59, meta: "Bad request: a fragment is not allowed" });
}
// `gemini://host` with nothing after the authority addresses the root.
let path = if path.is_empty() { "/" } else { path };
Ok(Request { host, path })
}
/// A status line and nothing else. `meta` is a redirect target for 3x and a
/// human-readable reason otherwise.
fn header(stream: &mut TcpStream, status: u8, meta: &str) -> Result<()> {
write!(stream, "{status} {meta}\r\n")?;
Ok(())
}
/// A 20 and a body. The redirect targets this listener emits are relative
/// references, which the spec permits and which are the only correct form here:
/// the port this socket is bound to is the terminator's back end, not the port a
/// client reached, so an absolute URL built from it would send clients to a port
/// that is not published.
fn body_response(stream: &mut TcpStream, media_type: &str, body: &[u8]) -> Result<()> {
write!(stream, "20 {media_type}\r\n")?;
stream.write_all(body)?;
Ok(())
}
#[cfg(test)]
mod tests {
use super::*;
fn ok(line: &str) -> (String, String) {
let request = parse(line).unwrap_or_else(|_| panic!("{line} should parse"));
(request.host.to_string(), request.path.to_string())
}
fn refused(line: &str) -> u8 {
parse(line).err().unwrap_or_else(|| panic!("{line} should be refused")).status
}
#[test]
fn splits_an_ordinary_request() {
assert_eq!(
ok("gemini://example.org/notes/one"),
("example.org".into(), "/notes/one".into())
);
assert_eq!(ok("gemini://example.org/"), ("example.org".into(), "/".into()));
}
#[test]
fn a_bare_authority_addresses_the_root() {
assert_eq!(ok("gemini://example.org"), ("example.org".into(), "/".into()));
}
#[test]
fn keeps_the_port_on_the_host_for_normalisation_to_strip() {
assert_eq!(ok("gemini://example.org:1965/"), ("example.org:1965".into(), "/".into()));
}
#[test]
fn keeps_the_query_for_path_cleaning_to_drop() {
assert_eq!(
ok("gemini://example.org/?format=text"),
("example.org".into(), "/?format=text".into())
);
// A query with no path at all still leaves the path non-empty.
assert_eq!(ok("gemini://example.org?q=1"), ("example.org".into(), "?q=1".into()));
}
#[test]
fn the_scheme_is_matched_without_regard_to_case() {
assert_eq!(ok("GEMINI://example.org/").0, "example.org");
}
#[test]
fn another_scheme_is_a_proxy_request_not_a_parse_failure() {
// smolweb answers 59 here, which tells a client its request was malformed
// when it was merely for somewhere this server does not fetch from.
assert_eq!(refused("https://example.org/"), 53);
assert_eq!(refused("gopher://example.org/"), 53);
}
#[test]
fn a_relative_or_schemeless_request_is_a_bad_request() {
assert_eq!(refused("/notes/one"), 59);
assert_eq!(refused("example.org/notes"), 59);
assert_eq!(refused(""), 59);
}
#[test]
fn userinfo_and_fragments_are_bad_requests() {
// Both are forbidden by the spec rather than simply unsupported.
assert_eq!(refused("gemini://user@example.org/"), 59);
assert_eq!(refused("gemini://example.org/page#section"), 59);
assert_eq!(refused("gemini://example.org#section"), 59);
}
#[test]
fn an_empty_authority_is_a_bad_request() {
assert_eq!(refused("gemini:///notes"), 59);
}
}