kumo_wrap/
lib.rs

1use bstr::{BStr, ByteVec};
2
3pub const SOFT_WIDTH: usize = 75;
4pub const HARD_WIDTH: usize = 900;
5
6pub fn wrap(value: &str) -> String {
7    String::from_utf8(wrap_impl(value, SOFT_WIDTH, HARD_WIDTH)).expect("utf8-in, utf8-out")
8}
9
10pub fn wrap_bytes(value: impl AsRef<BStr>) -> Vec<u8> {
11    wrap_impl(value, SOFT_WIDTH, HARD_WIDTH)
12}
13
14/// We can't use textwrap::fill here because it will prefer to break
15/// a line rather than finding stuff that fits.  We use a simple
16/// algorithm that tries to fill up to the desired width, allowing
17/// for overflow if there is a word that is too long to fit in
18/// the header, but breaking after a hard limit threshold.
19pub fn wrap_impl(value: impl AsRef<BStr>, soft_width: usize, hard_width: usize) -> Vec<u8> {
20    let value: &BStr = value.as_ref();
21    let mut result: Vec<u8> = vec![];
22    let mut line: Vec<u8> = vec![];
23
24    for word in value.split(|&b| b.is_ascii_whitespace()) {
25        if word.is_empty() {
26            continue;
27        }
28        if line.len() + word.len() < soft_width {
29            if !line.is_empty() {
30                line.push(b' ');
31            }
32            line.push_str(word);
33            continue;
34        }
35
36        // Need to wrap.
37
38        // Accumulate line so far, if any
39        if !line.is_empty() {
40            if !result.is_empty() {
41                // There's an existing line, start a new one, indented
42                result.push(b'\t');
43            }
44            result.push_str(&line);
45            result.push_str("\r\n");
46            line.clear();
47        }
48
49        // build out a line from the characters of this word. `word` may contain
50        // multi-byte UTF-8 sequences (eg. an RFC 6531 addr-spec, which is
51        // emitted as raw UTF-8 rather than encoded-word wrapped), so
52        // hard-wrapping must cut on char boundaries to avoid producing invalid
53        // UTF-8. `word` isn't guaranteed to be valid UTF-8 (eg. it may come
54        // from an 8-bit header value), so we walk it in utf8_chunks and only
55        // split within the valid stretches. Any invalid byte run is pushed
56        // through unchanged rather than lossily replaced.
57        if word.len() <= hard_width {
58            line.push_str(word);
59        } else {
60            for chunk in word.utf8_chunks() {
61                for c in chunk.valid().chars() {
62                    let mut buf = [0u8; 4];
63                    line.push_str(c.encode_utf8(&mut buf).as_bytes());
64                    if line.len() >= hard_width {
65                        if !result.is_empty() {
66                            result.push(b'\t');
67                        }
68                        result.push_str(&line);
69                        result.push_str("\r\n");
70                        line.clear();
71                    }
72                }
73                if !chunk.invalid().is_empty() {
74                    line.push_str(chunk.invalid());
75                    if line.len() >= hard_width {
76                        if !result.is_empty() {
77                            result.push(b'\t');
78                        }
79                        result.push_str(&line);
80                        result.push_str("\r\n");
81                        line.clear();
82                    }
83                }
84            }
85        }
86    }
87
88    if !line.is_empty() {
89        if !result.is_empty() {
90            result.push(b'\t');
91        }
92        result.push_str(&line);
93    }
94
95    result
96}
97
98#[cfg(test)]
99mod test {
100    use super::*;
101
102    #[test]
103    fn wrapping() {
104        for (input, expect) in [
105            ("foo", "foo"),
106            ("hi there", "hi there"),
107            ("hello world", "hello\r\n\tworld"),
108            ("hello world ", "hello\r\n\tworld"),
109            (
110                "hello world foo bar baz woot woot",
111                "hello\r\n\tworld foo\r\n\tbar baz\r\n\twoot woot",
112            ),
113            (
114                "hi there breakmepleaseIamtoolong",
115                "hi there\r\n\tbreakmepleaseIa\r\n\tmtoolong",
116            ),
117        ] {
118            let wrapped = wrap_impl(input, 10, 15);
119            k9::assert_equal!(
120                wrapped,
121                expect.as_bytes(),
122                "input: '{input}' should produce '{expect}'"
123            );
124        }
125    }
126
127    /// A multi-byte word past the hard limit must split on char boundaries.
128    /// The 4-byte char with a hard limit of 15 (not a multiple of 4) forces
129    /// the wrap point inside a character, where byte-by-byte wrapping would
130    /// emit invalid UTF-8; the from_utf8 below is the check that catches it.
131    #[test]
132    fn hard_wrap_multibyte_word() {
133        let word = "😀".repeat(10); // 40 bytes, no ascii whitespace to fold at
134        let wrapped = wrap_impl(word.as_str(), 10, 15);
135        let text = String::from_utf8(wrapped).expect("wrapped output is valid UTF-8");
136        k9::assert_equal!(text.replace(['\r', '\n', '\t'], ""), word);
137    }
138
139    /// wrap() validates its output as UTF-8 via expect(); a word past the hard
140    /// limit whose characters do not align to it (the leading ASCII byte
141    /// offsets them) would, without char-boundary splitting, make that
142    /// validation panic.
143    #[test]
144    fn wrap_does_not_panic_on_long_multibyte_word() {
145        let word = format!("x{}", "😀".repeat(300)); // 1 + 1200 bytes, misaligned
146        let wrapped = wrap(&word);
147        k9::assert_equal!(wrapped.replace(['\r', '\n', '\t'], ""), word);
148    }
149
150    /// An over-long word carrying invalid UTF-8 keeps those bytes verbatim:
151    /// the hard-wrap walks utf8_chunks and must emit each invalid run as-is
152    /// rather than dropping it or substituting the replacement character.
153    /// Reachable only via wrap_bytes, since wrap() requires valid UTF-8.
154    #[test]
155    fn hard_wrap_preserves_invalid_utf8() {
156        let mut word = b"abc".to_vec();
157        word.extend_from_slice(&[0xff, 0xfe]); // invalid UTF-8 run
158        word.extend_from_slice(b"defghijklmnopqrstuv"); // pad past the hard limit
159        let wrapped = wrap_impl(word.as_slice(), 10, 15);
160        // Removing the inserted fold bytes must reconstruct the input exactly,
161        // invalid bytes included; the input has no CR/LF/TAB of its own.
162        let stripped: Vec<u8> = wrapped
163            .into_iter()
164            .filter(|b| !matches!(b, b'\r' | b'\n' | b'\t'))
165            .collect();
166        k9::assert_equal!(stripped, word);
167    }
168}