itsybitsy/core/src/wrap.rs

152 lines
4.8 KiB
Rust
Raw Normal View History

2026-10-04 20:27:29 +03:00
//! Measuring and wrapping text for the fixed-width formats.
//!
//! Shared rather than per-format so that a table's column alignment and a
//! paragraph's line breaks agree about how wide a character is. Width is counted
//! in terminal columns, not bytes or scalar values: a CJK ideograph occupies two
//! columns, and a combining mark none.
use unicode_width::{UnicodeWidthChar, UnicodeWidthStr};
/// Columns `text` occupies when printed.
pub fn display_width(text: &str) -> usize {
UnicodeWidthStr::width(text)
}
/// Pad `text` on the right to `width` columns.
pub fn pad(text: &str, width: usize) -> String {
let mut out = text.to_string();
for _ in display_width(text)..width {
out.push(' ');
}
out
}
/// Break `text` into lines no wider than `width` columns.
///
/// Greedy: each line takes as many words as fit. A word wider than the whole
/// width is split rather than left to overflow, since the formats this serves
/// have no horizontal scroll. A width of zero means do not wrap.
pub fn wrap(text: &str, width: usize) -> Vec<String> {
let text = text.trim();
if width == 0 {
return if text.is_empty() { Vec::new() } else { vec![text.to_string()] };
}
let mut lines = Vec::new();
let mut line = String::new();
let mut line_width = 0;
for word in text.split_whitespace() {
for piece in split_to_fit(word, width) {
let piece_width = display_width(&piece);
// The +1 is the space that would join it to what is already there.
if line_width > 0 && line_width + 1 + piece_width > width {
lines.push(std::mem::take(&mut line));
line_width = 0;
}
if line_width > 0 {
line.push(' ');
line_width += 1;
}
line.push_str(&piece);
line_width += piece_width;
}
}
if !line.is_empty() {
lines.push(line);
}
lines
}
/// Split one word into chunks that each fit `width`, on column boundaries.
///
/// A word that already fits comes back whole, which is the common case.
fn split_to_fit(word: &str, width: usize) -> Vec<String> {
if display_width(word) <= width {
return vec![word.to_string()];
}
let mut chunks = Vec::new();
let mut chunk = String::new();
let mut chunk_width = 0;
for ch in word.chars() {
let ch_width = UnicodeWidthChar::width(ch).unwrap_or(0);
if chunk_width + ch_width > width && !chunk.is_empty() {
chunks.push(std::mem::take(&mut chunk));
chunk_width = 0;
}
chunk.push(ch);
chunk_width += ch_width;
}
if !chunk.is_empty() {
chunks.push(chunk);
}
chunks
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn measures_in_terminal_columns() {
assert_eq!(display_width("abc"), 3);
// A CJK ideograph occupies two columns, not one character's worth.
assert_eq!(display_width("日本語"), 6);
// A combining mark occupies none.
assert_eq!(display_width("e\u{301}"), 1);
}
#[test]
fn pads_to_a_column_count() {
assert_eq!(pad("ab", 5), "ab ");
assert_eq!(pad("日本", 6), "日本 ");
// Already at or over the width, so nothing is added.
assert_eq!(pad("abcde", 5), "abcde");
assert_eq!(pad("abcdef", 5), "abcdef");
}
#[test]
fn fills_each_line_greedily() {
assert_eq!(wrap("one two three four", 9), vec!["one two", "three", "four"]);
assert_eq!(wrap("aaa bbb", 7), vec!["aaa bbb"], "an exact fit stays on one line");
}
#[test]
fn collapses_runs_of_whitespace() {
assert_eq!(wrap("a b\n\tc", 80), vec!["a b c"]);
}
#[test]
fn a_word_wider_than_the_line_is_split_rather_than_overflowing() {
// These formats have no horizontal scroll, so overflow would be lost.
assert_eq!(wrap("aaaaaaaa", 3), vec!["aaa", "aaa", "aa"]);
assert_eq!(wrap("ok aaaaa", 3), vec!["ok", "aaa", "aa"]);
}
#[test]
fn a_split_never_lands_mid_column() {
// Splitting between the two columns of a wide character would corrupt it.
assert_eq!(wrap("日本語", 3), vec!["日", "本", "語"]);
assert_eq!(wrap("日本語", 4), vec!["日本", "語"]);
}
#[test]
fn a_width_of_zero_means_do_not_wrap() {
assert_eq!(wrap("one two three", 0), vec!["one two three"]);
assert!(wrap("", 0).is_empty());
}
#[test]
fn empty_input_produces_no_lines() {
assert!(wrap("", 20).is_empty());
assert!(wrap(" \n ", 20).is_empty());
}
#[test]
fn no_line_carries_trailing_whitespace() {
for line in wrap("one two three four five six", 10) {
assert_eq!(line.trim_end(), line, "{line:?}");
}
}
}