//! Measuring and wrapping text for the fixed-width formats. //! //! Shared rather than per-format so that a table's column alignment and a //! paragraph's line breaks agree about how wide a character is. Width is counted //! in terminal columns, not bytes or scalar values: a CJK ideograph occupies two //! columns, and a combining mark none. use unicode_width::{UnicodeWidthChar, UnicodeWidthStr}; /// Columns `text` occupies when printed. pub fn display_width(text: &str) -> usize { UnicodeWidthStr::width(text) } /// Pad `text` on the right to `width` columns. pub fn pad(text: &str, width: usize) -> String { let mut out = text.to_string(); for _ in display_width(text)..width { out.push(' '); } out } /// Break `text` into lines no wider than `width` columns. /// /// Greedy: each line takes as many words as fit. A word wider than the whole /// width is split rather than left to overflow, since the formats this serves /// have no horizontal scroll. A width of zero means do not wrap. pub fn wrap(text: &str, width: usize) -> Vec { let text = text.trim(); if width == 0 { return if text.is_empty() { Vec::new() } else { vec![text.to_string()] }; } let mut lines = Vec::new(); let mut line = String::new(); let mut line_width = 0; for word in text.split_whitespace() { for piece in split_to_fit(word, width) { let piece_width = display_width(&piece); // The +1 is the space that would join it to what is already there. if line_width > 0 && line_width + 1 + piece_width > width { lines.push(std::mem::take(&mut line)); line_width = 0; } if line_width > 0 { line.push(' '); line_width += 1; } line.push_str(&piece); line_width += piece_width; } } if !line.is_empty() { lines.push(line); } lines } /// Split one word into chunks that each fit `width`, on column boundaries. /// /// A word that already fits comes back whole, which is the common case. fn split_to_fit(word: &str, width: usize) -> Vec { if display_width(word) <= width { return vec![word.to_string()]; } let mut chunks = Vec::new(); let mut chunk = String::new(); let mut chunk_width = 0; for ch in word.chars() { let ch_width = UnicodeWidthChar::width(ch).unwrap_or(0); if chunk_width + ch_width > width && !chunk.is_empty() { chunks.push(std::mem::take(&mut chunk)); chunk_width = 0; } chunk.push(ch); chunk_width += ch_width; } if !chunk.is_empty() { chunks.push(chunk); } chunks } #[cfg(test)] mod tests { use super::*; #[test] fn measures_in_terminal_columns() { assert_eq!(display_width("abc"), 3); // A CJK ideograph occupies two columns, not one character's worth. assert_eq!(display_width("日本語"), 6); // A combining mark occupies none. assert_eq!(display_width("e\u{301}"), 1); } #[test] fn pads_to_a_column_count() { assert_eq!(pad("ab", 5), "ab "); assert_eq!(pad("日本", 6), "日本 "); // Already at or over the width, so nothing is added. assert_eq!(pad("abcde", 5), "abcde"); assert_eq!(pad("abcdef", 5), "abcdef"); } #[test] fn fills_each_line_greedily() { assert_eq!(wrap("one two three four", 9), vec!["one two", "three", "four"]); assert_eq!(wrap("aaa bbb", 7), vec!["aaa bbb"], "an exact fit stays on one line"); } #[test] fn collapses_runs_of_whitespace() { assert_eq!(wrap("a b\n\tc", 80), vec!["a b c"]); } #[test] fn a_word_wider_than_the_line_is_split_rather_than_overflowing() { // These formats have no horizontal scroll, so overflow would be lost. assert_eq!(wrap("aaaaaaaa", 3), vec!["aaa", "aaa", "aa"]); assert_eq!(wrap("ok aaaaa", 3), vec!["ok", "aaa", "aa"]); } #[test] fn a_split_never_lands_mid_column() { // Splitting between the two columns of a wide character would corrupt it. assert_eq!(wrap("日本語", 3), vec!["日", "本", "語"]); assert_eq!(wrap("日本語", 4), vec!["日本", "語"]); } #[test] fn a_width_of_zero_means_do_not_wrap() { assert_eq!(wrap("one two three", 0), vec!["one two three"]); assert!(wrap("", 0).is_empty()); } #[test] fn empty_input_produces_no_lines() { assert!(wrap("", 20).is_empty()); assert!(wrap(" \n ", 20).is_empty()); } #[test] fn no_line_carries_trailing_whitespace() { for line in wrap("one two three four five six", 10) { assert_eq!(line.trim_end(), line, "{line:?}"); } } }