kumo_wrap/lib.rs
1use bstr::{BStr, ByteVec};
2
3pub const SOFT_WIDTH: usize = 75;
4pub const HARD_WIDTH: usize = 900;
5
6pub fn wrap(value: &str) -> String {
7 String::from_utf8(wrap_impl(value, SOFT_WIDTH, HARD_WIDTH)).expect("utf8-in, utf8-out")
8}
9
10pub fn wrap_bytes(value: impl AsRef<BStr>) -> Vec<u8> {
11 wrap_impl(value, SOFT_WIDTH, HARD_WIDTH)
12}
13
14/// We can't use textwrap::fill here because it will prefer to break
15/// a line rather than finding stuff that fits. We use a simple
16/// algorithm that tries to fill up to the desired width, allowing
17/// for overflow if there is a word that is too long to fit in
18/// the header, but breaking after a hard limit threshold.
19pub fn wrap_impl(value: impl AsRef<BStr>, soft_width: usize, hard_width: usize) -> Vec<u8> {
20 let value: &BStr = value.as_ref();
21 let mut result: Vec<u8> = vec![];
22 let mut line: Vec<u8> = vec![];
23
24 for word in value.split(|&b| b.is_ascii_whitespace()) {
25 if word.is_empty() {
26 continue;
27 }
28 if line.len() + word.len() < soft_width {
29 if !line.is_empty() {
30 line.push(b' ');
31 }
32 line.push_str(word);
33 continue;
34 }
35
36 // Need to wrap.
37
38 // Accumulate line so far, if any
39 if !line.is_empty() {
40 if !result.is_empty() {
41 // There's an existing line, start a new one, indented
42 result.push(b'\t');
43 }
44 result.push_str(&line);
45 result.push_str("\r\n");
46 line.clear();
47 }
48
49 // build out a line from the characters of this word. `word` may contain
50 // multi-byte UTF-8 sequences (eg. an RFC 6531 addr-spec, which is
51 // emitted as raw UTF-8 rather than encoded-word wrapped), so
52 // hard-wrapping must cut on char boundaries to avoid producing invalid
53 // UTF-8. `word` isn't guaranteed to be valid UTF-8 (eg. it may come
54 // from an 8-bit header value), so we walk it in utf8_chunks and only
55 // split within the valid stretches. Any invalid byte run is pushed
56 // through unchanged rather than lossily replaced.
57 if word.len() <= hard_width {
58 line.push_str(word);
59 } else {
60 for chunk in word.utf8_chunks() {
61 for c in chunk.valid().chars() {
62 let mut buf = [0u8; 4];
63 line.push_str(c.encode_utf8(&mut buf).as_bytes());
64 if line.len() >= hard_width {
65 if !result.is_empty() {
66 result.push(b'\t');
67 }
68 result.push_str(&line);
69 result.push_str("\r\n");
70 line.clear();
71 }
72 }
73 if !chunk.invalid().is_empty() {
74 line.push_str(chunk.invalid());
75 if line.len() >= hard_width {
76 if !result.is_empty() {
77 result.push(b'\t');
78 }
79 result.push_str(&line);
80 result.push_str("\r\n");
81 line.clear();
82 }
83 }
84 }
85 }
86 }
87
88 if !line.is_empty() {
89 if !result.is_empty() {
90 result.push(b'\t');
91 }
92 result.push_str(&line);
93 }
94
95 result
96}
97
98#[cfg(test)]
99mod test {
100 use super::*;
101
102 #[test]
103 fn wrapping() {
104 for (input, expect) in [
105 ("foo", "foo"),
106 ("hi there", "hi there"),
107 ("hello world", "hello\r\n\tworld"),
108 ("hello world ", "hello\r\n\tworld"),
109 (
110 "hello world foo bar baz woot woot",
111 "hello\r\n\tworld foo\r\n\tbar baz\r\n\twoot woot",
112 ),
113 (
114 "hi there breakmepleaseIamtoolong",
115 "hi there\r\n\tbreakmepleaseIa\r\n\tmtoolong",
116 ),
117 ] {
118 let wrapped = wrap_impl(input, 10, 15);
119 k9::assert_equal!(
120 wrapped,
121 expect.as_bytes(),
122 "input: '{input}' should produce '{expect}'"
123 );
124 }
125 }
126
127 /// A multi-byte word past the hard limit must split on char boundaries.
128 /// The 4-byte char with a hard limit of 15 (not a multiple of 4) forces
129 /// the wrap point inside a character, where byte-by-byte wrapping would
130 /// emit invalid UTF-8; the from_utf8 below is the check that catches it.
131 #[test]
132 fn hard_wrap_multibyte_word() {
133 let word = "😀".repeat(10); // 40 bytes, no ascii whitespace to fold at
134 let wrapped = wrap_impl(word.as_str(), 10, 15);
135 let text = String::from_utf8(wrapped).expect("wrapped output is valid UTF-8");
136 k9::assert_equal!(text.replace(['\r', '\n', '\t'], ""), word);
137 }
138
139 /// wrap() validates its output as UTF-8 via expect(); a word past the hard
140 /// limit whose characters do not align to it (the leading ASCII byte
141 /// offsets them) would, without char-boundary splitting, make that
142 /// validation panic.
143 #[test]
144 fn wrap_does_not_panic_on_long_multibyte_word() {
145 let word = format!("x{}", "😀".repeat(300)); // 1 + 1200 bytes, misaligned
146 let wrapped = wrap(&word);
147 k9::assert_equal!(wrapped.replace(['\r', '\n', '\t'], ""), word);
148 }
149
150 /// An over-long word carrying invalid UTF-8 keeps those bytes verbatim:
151 /// the hard-wrap walks utf8_chunks and must emit each invalid run as-is
152 /// rather than dropping it or substituting the replacement character.
153 /// Reachable only via wrap_bytes, since wrap() requires valid UTF-8.
154 #[test]
155 fn hard_wrap_preserves_invalid_utf8() {
156 let mut word = b"abc".to_vec();
157 word.extend_from_slice(&[0xff, 0xfe]); // invalid UTF-8 run
158 word.extend_from_slice(b"defghijklmnopqrstuv"); // pad past the hard limit
159 let wrapped = wrap_impl(word.as_slice(), 10, 15);
160 // Removing the inserted fold bytes must reconstruct the input exactly,
161 // invalid bytes included; the input has no CR/LF/TAB of its own.
162 let stripped: Vec<u8> = wrapped
163 .into_iter()
164 .filter(|b| !matches!(b, b'\r' | b'\n' | b'\t'))
165 .collect();
166 k9::assert_equal!(stripped, word);
167 }
168}