fix: join the RNS loop thread before exit and keep the response slot per-request

This commit is contained in:
randogoth 2026-09-29 09:59:45 +03:00
parent 39c4c4d6a5
commit 234b75651f
17 changed files with 84 additions and 11 deletions

View file

@ -41,6 +41,10 @@ static RNS::Link active_link({RNS::Type::NONE});
static microStore::FileSystem filesystem{microStore::Adapters::UniversalFileSystem()};
static volatile bool running = false;
// Joinable, never detached: the loop thread must be joined by
// smolmail_rns_stop before the process tears down statics, or it keeps
// calling reticulum.loop() while their destructors run.
static std::thread loop_thread;
// The single response slot (plan: one request in flight at a time).
static std::mutex slot_mutex;
@ -94,9 +98,11 @@ static void on_failed(const RNS::RequestReceipt& receipt) {
slot_cv.notify_all();
}
// The unwrapped large-response payload, filled in by smolmail_rns_request.
static RNS::Bytes unwrapped;
// The unwrapped large-response payload, filled in per request below. It is
// deliberately local: a static here kept the last resource-path response
// alive past its request, so every later bare response -- an empty fetch
// page, an ack -- was misread as that stale page and the client looped
// FETCH/DELETE pairs forever.
static void loop_thread_main() {
while (running) {
reticulum.loop();
@ -139,7 +145,7 @@ extern "C" int smolmail_rns_start(const char* storage_dir,
reticulum.start();
running = true;
std::thread(loop_thread_main).detach();
loop_thread = std::thread(loop_thread_main);
return 0;
}
@ -256,6 +262,7 @@ extern "C" int smolmail_rns_request(const uint8_t* request, size_t request_len,
// raw remainder to handle_response on that path). The wrapper is
// unambiguous: a smolmail status is a single byte below 16, and the
// msgpack bin headers are 0xC4..0xC6, so only those are unwrapped.
RNS::Bytes unwrapped;
const RNS::Bytes& payload = [&]() -> const RNS::Bytes& {
if (slot_response.size() > 0 && slot_response.data()[0] >= 0xC4
&& slot_response.data()[0] <= 0xC6) {
@ -286,3 +293,23 @@ extern "C" void smolmail_rns_close(void) {
active_link = RNS::Link({RNS::Type::NONE});
}
}
extern "C" void smolmail_rns_stop(void) {
if (!running) {
return;
}
// Order matters: halt and join the loop thread first, so nothing is
// inside Reticulum, Transport or the filesystem while they are torn
// down; only then take the link, the interface and the instance apart.
running = false;
if (loop_thread.joinable()) {
loop_thread.join();
}
if (active_link) {
active_link.teardown();
active_link = RNS::Link({RNS::Type::NONE});
}
RNS::Transport::deregister_interface(udp_interface);
udp_interface.stop();
reticulum = RNS::Reticulum({RNS::Type::NONE});
}

View file

@ -52,6 +52,14 @@ int smolmail_rns_request(const uint8_t *request, size_t request_len,
* minutes (upstream spec sec 13.10). */
void smolmail_rns_close(void);
/* Stops the Reticulum stack: halts and joins the loop thread, tears the link
* down, deregisters and stops the UDP interface, and releases the
* Reticulum instance. The loop thread must be joined before the process (or
* embedding host) tears down statics, or it keeps calling into Reticulum
* while their destructors run. A no-op when the stack is not running.
* The stack can be started again afterwards. */
void smolmail_rns_stop(void);
#ifdef __cplusplus
}
#endif

View file

@ -42,6 +42,8 @@ mod inner {
) -> c_int;
pub fn smolmail_rns_close();
pub fn smolmail_rns_stop();
}
}
@ -150,6 +152,14 @@ pub fn close() {
unsafe { inner::smolmail_rns_close() };
}
/// Stops the Reticulum stack: joins the loop thread, tears the link down,
/// stops the UDP interface and releases the instance. Must run before the
/// embedding process tears down statics; a no-op when never started, and
/// the stack can be started again afterwards.
pub fn stop() {
unsafe { inner::smolmail_rns_stop() };
}
/// The destination hash length, for callers checking address shapes.
pub const DEST_LEN: usize = 16;
const _: () = assert!(DEST_LEN == 16 && KEY_LEN == 32);

View file

@ -2,7 +2,10 @@
//! the C shim, dispatched through the same operations as TCP.
//!
//! The carrier starts lazily, on the first `smol+rns://` dial, so local-only
//! commands stay usable offline and start instantly. A client creates no
//! commands stay usable offline and start instantly, and stops via `stop()`,
//! which every embedding process must call before it tears down: the
//! Reticulum loop thread otherwise races the destruction of the statics it
//! is calling into. A client creates no
//! Reticulum identity (upstream spec sec 13.8): the shim never calls
//! `Link::identify`, and the flags below configure only the storage path and
//! the UDP interface.
@ -11,6 +14,7 @@ pub mod ffi;
pub mod transport;
use std::path::Path;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Mutex, OnceLock};
use crate::error::Error;
@ -44,7 +48,7 @@ impl Default for RnsConfig {
}
static CONFIG: OnceLock<RnsConfig> = OnceLock::new();
static STARTED: OnceLock<()> = OnceLock::new();
static STARTED: AtomicBool = AtomicBool::new(false);
static START_MUTEX: Mutex<()> = Mutex::new(());
/// Records the carrier configuration; the first dial wins, as in a CLI the
@ -53,19 +57,33 @@ pub fn configure(config: RnsConfig) {
let _ = CONFIG.set(config);
}
/// Starts the Reticulum stack at most once, on the first RNS dial.
/// Starts the Reticulum stack at most once between calls to `stop`, on the
/// first RNS dial.
pub fn ensure_started() -> Result<(), Error> {
if STARTED.get().is_some() {
if STARTED.load(Ordering::Acquire) {
return Ok(());
}
let _guard = START_MUTEX.lock().unwrap();
if STARTED.get().is_some() {
if STARTED.load(Ordering::Acquire) {
return Ok(());
}
let config = CONFIG.get().cloned().unwrap_or_default();
std::fs::create_dir_all(Path::new(&config.storage_dir))
.map_err(|e| Error::Other(format!("cannot create {}: {e}", config.storage_dir)))?;
ffi::start(&config)?;
let _ = STARTED.set(());
STARTED.store(true, Ordering::Release);
Ok(())
}
/// Stops the Reticulum stack and joins its loop thread. The shim's loop
/// thread otherwise outlives the caller and races the teardown of its statics
/// at process exit — a nondeterministic hang the RNS carrier hit on a
/// first-run provisioning pass. Embedders call this when they are done with
/// the carrier; the CLI calls it before exiting. A no-op when never started,
/// and the stack starts again on the next dial.
pub fn stop() {
if !STARTED.swap(false, Ordering::AcqRel) {
return;
}
ffi::stop();
}