mailparsing/
rfc5322_parser.rs

1use crate::headermap::EncodeHeaderValue;
2use crate::{MailParsingError, Result, SharedString};
3use bstr::{BStr, BString, ByteSlice, ByteVec};
4use charset_normalizer_rs::Encoding;
5use nom::branch::alt;
6use nom::bytes::complete::{take_while, take_while1, take_while_m_n};
7use nom::combinator::{all_consuming, map, opt, recognize};
8use nom::error::context;
9use nom::multi::{many0, many1, separated_list1};
10use nom::sequence::{delimited, preceded, separated_pair, terminated};
11use nom::Parser as _;
12use nom_utils::{
13    explain_nom, make_context_error, make_span, tag, utf8_non_ascii, IResult, ParseError, Span,
14};
15use serde::{Deserialize, Serialize};
16use serde_with::{serde_as, DeserializeAs, SerializeAs};
17use std::collections::{BTreeMap, BTreeSet};
18use std::fmt::Debug;
19
20/// A `serde_with` adapter that serializes `BString` as a JSON string when
21/// the value is valid UTF-8, falling back to the default byte-array
22/// representation otherwise.
23pub struct BStringUtf8;
24
25impl SerializeAs<BString> for BStringUtf8 {
26    fn serialize_as<S>(value: &BString, serializer: S) -> std::result::Result<S::Ok, S::Error>
27    where
28        S: serde::Serializer,
29    {
30        match std::str::from_utf8(value.as_bytes()) {
31            Ok(s) => serializer.serialize_str(s),
32            Err(_) => value.serialize(serializer),
33        }
34    }
35}
36
37impl<'de> DeserializeAs<'de, BString> for BStringUtf8 {
38    fn deserialize_as<D>(deserializer: D) -> std::result::Result<BString, D::Error>
39    where
40        D: serde::Deserializer<'de>,
41    {
42        BString::deserialize(deserializer)
43    }
44}
45
46impl MailParsingError {
47    pub fn from_nom(input: Span, err: nom::Err<ParseError<Span<'_>>>) -> Self {
48        MailParsingError::HeaderParse(explain_nom(input, err))
49    }
50}
51
52// ctl = { '\u{00}'..'\u{1f}' | "\u{7f}" }
53fn is_ctl(c: u8) -> bool {
54    match c {
55        b'\x00'..=b'\x1f' | b'\x7f' => true,
56        _ => false,
57    }
58}
59
60fn not_angle(c: u8) -> bool {
61    match c {
62        b'<' | b'>' => false,
63        _ => true,
64    }
65}
66
67// char = { '\u{01}'..'\u{7f}' }
68fn is_char(c: u8) -> bool {
69    match c {
70        0x01..=0x7f => true,
71        _ => false,
72    }
73}
74
75fn is_especial(c: u8) -> bool {
76    match c {
77        b'(' | b')' | b'<' | b'>' | b'@' | b',' | b';' | b':' | b'/' | b'[' | b']' | b'?'
78        | b'.' | b'=' => true,
79        _ => false,
80    }
81}
82
83fn is_token(c: u8) -> bool {
84    is_char(c) && c != b' ' && !is_especial(c) && !is_ctl(c)
85}
86
87// vchar = { '\u{21}'..'\u{7e}' | utf8_non_ascii }
88fn is_vchar_ascii(c: u8) -> bool {
89    (0x21..=0x7e).contains(&c)
90}
91
92fn is_atext_ascii(c: u8) -> bool {
93    match c {
94        b'!' | b'#' | b'$' | b'%' | b'&' | b'\'' | b'*' | b'+' | b'-' | b'/' | b'=' | b'?'
95        | b'^' | b'_' | b'`' | b'{' | b'|' | b'}' | b'~' => true,
96        c => c.is_ascii_alphanumeric(),
97    }
98}
99
100/// Byte-level predicate for atext, including UTF-8 continuation/leading bytes.
101/// Used for non-parser checks (e.g., needs_quoting). For parsing, use the
102/// `atext` parser which properly validates UTF-8 via `utf8_non_ascii`.
103fn is_atext(c: u8) -> bool {
104    is_atext_ascii(c) || c >= 0x80
105}
106
107fn atext(input: Span) -> IResult<Span, Span> {
108    context(
109        "atext",
110        recognize(many1(alt((take_while1(is_atext_ascii), utf8_non_ascii)))),
111    )
112    .parse(input)
113}
114
115fn is_obs_no_ws_ctl(c: u8) -> bool {
116    match c {
117        0x01..=0x08 | 0x0b..=0x0c | 0x0e..=0x1f | 0x7f => true,
118        _ => false,
119    }
120}
121
122fn is_obs_ctext(c: u8) -> bool {
123    is_obs_no_ws_ctl(c)
124}
125
126// ctext = { '\u{21}'..'\u{27}' | '\u{2a}'..'\u{5b}' | '\u{5d}'..'\u{7e}' | obs_ctext | utf8_non_ascii }
127fn is_ctext_ascii(c: u8) -> bool {
128    match c {
129        0x21..=0x27 | 0x2a..=0x5b | 0x5d..=0x7e => true,
130        c => is_obs_ctext(c),
131    }
132}
133
134// dtext = { '\u{21}'..'\u{5a}' | '\u{5e}'..'\u{7e}' | obs_dtext | utf8_non_ascii }
135// obs_dtext = { obs_no_ws_ctl | quoted_pair }
136fn is_dtext_ascii(c: u8) -> bool {
137    match c {
138        0x21..=0x5a | 0x5e..=0x7e => true,
139        c => is_obs_no_ws_ctl(c),
140    }
141}
142
143// qtext = { "\u{21}" | '\u{23}'..'\u{5b}' | '\u{5d}'..'\u{7e}' | obs_qtext | utf8_non_ascii }
144// obs_qtext = { obs_no_ws_ctl }
145fn is_qtext_ascii(c: u8) -> bool {
146    match c {
147        0x21 | 0x23..=0x5b | 0x5d..=0x7e => true,
148        c => is_obs_no_ws_ctl(c),
149    }
150}
151
152/// Byte-level predicate for qtext, including UTF-8 continuation/leading bytes.
153/// Used for non-parser checks. For parsing, use `qcontent` which validates
154/// UTF-8 via `utf8_non_ascii`.
155fn is_qtext(c: u8) -> bool {
156    is_qtext_ascii(c) || c >= 0x80
157}
158
159fn is_tspecial(c: u8) -> bool {
160    match c {
161        b'(' | b')' | b'<' | b'>' | b'@' | b',' | b';' | b':' | b'\\' | b'"' | b'/' | b'['
162        | b']' | b'?' | b'=' => true,
163        _ => false,
164    }
165}
166
167fn is_attribute_char(c: u8) -> bool {
168    match c {
169        b' ' | b'*' | b'\'' | b'%' => false,
170        _ => is_char(c) && !is_ctl(c) && !is_tspecial(c),
171    }
172}
173
174fn wsp(input: Span) -> IResult<Span, Span> {
175    context("wsp", take_while1(|c| c == b' ' || c == b'\t')).parse(input)
176}
177
178fn newline(input: Span) -> IResult<Span, Span> {
179    context("newline", recognize(preceded(opt(tag("\r")), tag("\n")))).parse(input)
180}
181
182// fws = { ((wsp* ~ "\r"? ~ "\n")* ~ wsp+) | obs_fws }
183fn fws(input: Span) -> IResult<Span, Span> {
184    context(
185        "fws",
186        alt((
187            recognize(preceded(many0(preceded(many0(wsp), newline)), many1(wsp))),
188            obs_fws,
189        )),
190    )
191    .parse(input)
192}
193
194// obs_fws = { wsp+ ~ ("\r"? ~ "\n" ~ wsp+)* }
195fn obs_fws(input: Span) -> IResult<Span, Span> {
196    context(
197        "obs_fws",
198        recognize(preceded(many1(wsp), preceded(newline, many1(wsp)))),
199    )
200    .parse(input)
201}
202
203// mailbox_list = { (mailbox ~ ("," ~ mailbox)*) | obs_mbox_list }
204fn mailbox_list(input: Span) -> IResult<Span, MailboxList> {
205    let (loc, mailboxes) = context(
206        "mailbox_list",
207        alt((separated_list1(tag(","), mailbox), obs_mbox_list)),
208    )
209    .parse(input)?;
210    Ok((loc, MailboxList(mailboxes)))
211}
212
213// obs_mbox_list = {  ((cfws? ~ ",")* ~ mailbox ~ ("," ~ (mailbox | cfws))*)+ }
214fn obs_mbox_list(input: Span) -> IResult<Span, Vec<Mailbox>> {
215    let (loc, entries) = context(
216        "obs_mbox_list",
217        many1(preceded(
218            many0(preceded(opt(cfws), tag(","))),
219            (
220                mailbox,
221                many0(preceded(
222                    tag(","),
223                    alt((map(mailbox, Some), map(cfws, |_| None))),
224                )),
225            ),
226        )),
227    )
228    .parse(input)?;
229
230    let mut result: Vec<Mailbox> = vec![];
231
232    for (first, boxes) in entries {
233        result.push(first);
234        for b in boxes {
235            if let Some(m) = b {
236                result.push(m);
237            }
238        }
239    }
240
241    Ok((loc, result))
242}
243
244// mailbox = { name_addr | addr_spec }
245fn mailbox(input: Span) -> IResult<Span, Mailbox> {
246    if let Ok(res) = name_addr(input) {
247        Ok(res)
248    } else {
249        let (loc, address) = context("mailbox", addr_spec).parse(input)?;
250        Ok((
251            loc,
252            Mailbox {
253                name: None,
254                address,
255            },
256        ))
257    }
258}
259
260// address_list = { (address ~ ("," ~ address)*) | obs_addr_list }
261fn address_list(input: Span) -> IResult<Span, AddressList> {
262    context(
263        "address_list",
264        alt((
265            map(separated_list1(tag(","), address), AddressList),
266            obs_address_list,
267        )),
268    )
269    .parse(input)
270}
271
272// obs_addr_list = {  ((cfws? ~ ",")* ~ address ~ ("," ~ (address | cfws))*)+ }
273fn obs_address_list(input: Span) -> IResult<Span, AddressList> {
274    let (loc, entries) = context(
275        "obs_address_list",
276        many1(preceded(
277            many0(preceded(opt(cfws), tag(","))),
278            (
279                address,
280                many0(preceded(
281                    tag(","),
282                    alt((map(address, Some), map(cfws, |_| None))),
283                )),
284            ),
285        )),
286    )
287    .parse(input)?;
288
289    let mut result: Vec<Address> = vec![];
290
291    for (first, boxes) in entries {
292        result.push(first);
293        for b in boxes {
294            if let Some(m) = b {
295                result.push(m);
296            }
297        }
298    }
299
300    Ok((loc, AddressList(result)))
301}
302
303// address = { mailbox | group }
304fn address(input: Span) -> IResult<Span, Address> {
305    context("address", alt((map(mailbox, Address::Mailbox), group))).parse(input)
306}
307
308// group = { display_name ~ ":" ~ group_list? ~ ";" ~ cfws? }
309fn group(input: Span) -> IResult<Span, Address> {
310    let (loc, (name, _, group_list, _)) = context(
311        "group",
312        terminated(
313            (display_name, tag(":"), opt(group_list), tag(";")),
314            opt(cfws),
315        ),
316    )
317    .parse(input)?;
318    Ok((
319        loc,
320        Address::Group {
321            name,
322            entries: group_list.unwrap_or_else(|| MailboxList(vec![])),
323        },
324    ))
325}
326
327// group_list = { mailbox_list | cfws | obs_group_list }
328fn group_list(input: Span) -> IResult<Span, MailboxList> {
329    context(
330        "group_list",
331        alt((
332            mailbox_list,
333            map(cfws, |_| MailboxList(vec![])),
334            obs_group_list,
335        )),
336    )
337    .parse(input)
338}
339
340// obs_group_list = @{ (cfws? ~ ",")+ ~ cfws? }
341fn obs_group_list(input: Span) -> IResult<Span, MailboxList> {
342    context(
343        "obs_group_list",
344        map(
345            terminated(many1(preceded(opt(cfws), tag(","))), opt(cfws)),
346            |_| MailboxList(vec![]),
347        ),
348    )
349    .parse(input)
350}
351
352// name_addr = { display_name? ~ angle_addr }
353fn name_addr(input: Span) -> IResult<Span, Mailbox> {
354    context(
355        "name_addr",
356        map((opt(display_name), angle_addr), |(name, address)| Mailbox {
357            name,
358            address,
359        }),
360    )
361    .parse(input)
362}
363
364// display_name = { phrase }
365fn display_name(input: Span) -> IResult<Span, String> {
366    context("display_name", phrase).parse(input)
367}
368
369// phrase = { (encoded_word | word)+ | obs_phrase }
370// obs_phrase = { (encoded_word | word) ~ (encoded_word | word | dot | cfws)* }
371fn phrase(input: Span) -> IResult<Span, String> {
372    let (loc, (a, b)): (Span, (BString, Vec<Option<BString>>)) = context(
373        "phrase",
374        (
375            alt((encoded_word, word)),
376            many0(alt((
377                map(cfws, |_| None),
378                map(encoded_word, Option::Some),
379                map(word, Option::Some),
380                map(tag("."), |_dot| Some(BString::from("."))),
381            ))),
382        ),
383    )
384    .parse(input)?;
385    let mut result = a;
386    for item in b {
387        if let Some(item) = item {
388            result.push(b' ');
389            result.push_str(item);
390        }
391    }
392    // SAFETY: all sub-parsers (word, encoded_word) produce only
393    // validated UTF-8 via utf8_non_ascii or charset decoding.
394    Ok((
395        loc,
396        String::from_utf8(result.into())
397            .expect("phrase sub-parsers should only produce valid UTF-8"),
398    ))
399}
400
401// angle_addr = { cfws? ~ "<" ~ addr_spec ~ ">" ~ cfws? | obs_angle_addr }
402fn angle_addr(input: Span) -> IResult<Span, AddrSpec> {
403    context(
404        "angle_addr",
405        alt((
406            delimited(
407                opt(cfws),
408                delimited(tag("<"), addr_spec, tag(">")),
409                opt(cfws),
410            ),
411            obs_angle_addr,
412        )),
413    )
414    .parse(input)
415}
416
417// obs_angle_addr = { cfws? ~ "<" ~ obs_route ~ addr_spec ~ ">" ~ cfws? }
418fn obs_angle_addr(input: Span) -> IResult<Span, AddrSpec> {
419    context(
420        "obs_angle_addr",
421        delimited(
422            opt(cfws),
423            delimited(tag("<"), preceded(obs_route, addr_spec), tag(">")),
424            opt(cfws),
425        ),
426    )
427    .parse(input)
428}
429
430// obs_route = { obs_domain_list ~ ":" }
431// obs_domain_list = { (cfws | ",")* ~ "@" ~ domain ~ ("," ~ cfws? ~ ("@" ~ domain)?)* }
432fn obs_route(input: Span) -> IResult<Span, Span> {
433    context(
434        "obs_route",
435        recognize(terminated(
436            (
437                many0(alt((cfws, recognize(tag(","))))),
438                recognize(tag("@")),
439                recognize(domain),
440                many0((tag(","), opt(cfws), opt((tag("@"), domain)))),
441            ),
442            tag(":"),
443        )),
444    )
445    .parse(input)
446}
447
448// addr_spec = { local_part ~ "@" ~ domain }
449fn addr_spec(input: Span) -> IResult<Span, AddrSpec> {
450    let (loc, (local_part, domain)) =
451        context("addr_spec", separated_pair(local_part, tag("@"), domain)).parse(input)?;
452
453    // local_part and domain parsers accept only ASCII or validated
454    // UTF-8 (via utf8_non_ascii), so this conversion is infallible.
455    let to_string = |b: BString| -> String {
456        String::from_utf8(b.into())
457            .expect("local_part/domain parsers should only produce valid UTF-8")
458    };
459
460    Ok((
461        loc,
462        AddrSpec {
463            local_part: to_string(local_part),
464            domain: to_string(domain),
465        },
466    ))
467}
468
469fn parse_with<'a, R, F>(text: &'a [u8], parser: F) -> Result<R>
470where
471    F: Fn(Span<'a>) -> IResult<'a, Span<'a>, R>,
472{
473    let input = make_span(text);
474    let (_, result) = all_consuming(parser)
475        .parse(input)
476        .map_err(|err| MailParsingError::from_nom(input, err))?;
477    Ok(result)
478}
479
480#[cfg(test)]
481#[test]
482fn test_addr_spec() {
483    k9::snapshot!(
484        parse_with("darth.vader@a.galaxy.far.far.away".as_bytes(), addr_spec),
485        r#"
486Ok(
487    AddrSpec {
488        local_part: "darth.vader",
489        domain: "a.galaxy.far.far.away",
490    },
491)
492"#
493    );
494
495    k9::snapshot!(
496        parse_with(
497            "\"darth.vader\"@a.galaxy.far.far.away".as_bytes(),
498            addr_spec
499        ),
500        r#"
501Ok(
502    AddrSpec {
503        local_part: "darth.vader",
504        domain: "a.galaxy.far.far.away",
505    },
506)
507"#
508    );
509
510    k9::snapshot!(
511        parse_with(
512            "\"darth\".vader@a.galaxy.far.far.away".as_bytes(),
513            addr_spec
514        ),
515        r#"
516Ok(
517    AddrSpec {
518        local_part: "darth.vader",
519        domain: "a.galaxy.far.far.away",
520    },
521)
522"#
523    );
524
525    k9::snapshot!(
526        parse_with("a@[127.0.0.1]".as_bytes(), addr_spec),
527        r#"
528Ok(
529    AddrSpec {
530        local_part: "a",
531        domain: "[127.0.0.1]",
532    },
533)
534"#
535    );
536
537    k9::snapshot!(
538        parse_with("a@[IPv6::1]".as_bytes(), addr_spec),
539        r#"
540Ok(
541    AddrSpec {
542        local_part: "a",
543        domain: "[IPv6::1]",
544    },
545)
546"#
547    );
548}
549
550#[cfg(test)]
551#[test]
552fn test_obs_local_part_in_addr_spec() {
553    // obs-local-part = word *("." word) where word = atom / quoted-string
554    // This mixed form is defined in RFC 5322 §4.4 and is correctly parsed
555    // via obs_local_part which is tried first in the local_part alternation.
556    k9::snapshot!(
557        parse_with(r#""first".last@example.com"#.as_bytes(), addr_spec),
558        r#"
559Ok(
560    AddrSpec {
561        local_part: "first.last",
562        domain: "example.com",
563    },
564)
565"#
566    );
567    k9::snapshot!(
568        parse_with(r#"first."last"@example.com"#.as_bytes(), addr_spec),
569        r#"
570Ok(
571    AddrSpec {
572        local_part: "first.last",
573        domain: "example.com",
574    },
575)
576"#
577    );
578    k9::snapshot!(
579        parse_with(r#""first"."last"@example.com"#.as_bytes(), addr_spec),
580        r#"
581Ok(
582    AddrSpec {
583        local_part: "first.last",
584        domain: "example.com",
585    },
586)
587"#
588    );
589}
590
591#[cfg(test)]
592#[test]
593fn test_obs_local_part_encode_roundtrip() {
594    // When an obs-local-part is resolved to its semantic content and stored
595    // in an AddrSpec, encode_value should produce a valid RFC 5321 address.
596
597    // "first".last -> semantic content "first.last" -> encodes as dot-string
598    let addr = AddrSpec::new("first.last", "example.com");
599    k9::assert_equal!(addr.encode_value(), "first.last@example.com");
600
601    // "first second".last -> semantic content "first second.last" -> needs quoting
602    let addr = AddrSpec::new("first second.last", "example.com");
603    k9::assert_equal!(addr.encode_value(), r#""first second.last"@example.com"#);
604
605    // "first\"".last -> semantic content "first\".last" -> needs quoting with escaping
606    let addr = AddrSpec::new("first\".last", "example.com");
607    k9::assert_equal!(addr.encode_value(), r#""first\".last"@example.com"#);
608}
609
610#[cfg(test)]
611#[test]
612fn test_encode_folds_long_mailbox_display_name() {
613    let mailbox = Mailbox {
614        name: Some(
615            "The Honorable Regional Manager of the Northwestern Sales Territory Office".to_string(),
616        ),
617        address: AddrSpec::new("alex", "example.com"),
618    };
619    let encoded = mailbox.encode_value().to_string();
620    k9::snapshot!(
621        BString::from(encoded.clone()),
622        r#"
623"The Honorable Regional Manager of the Northwestern Sales Territory Office"\r
624\t<alex@example.com>
625"#
626    );
627    k9::assert_equal!(
628        Parser::parse_mailbox_header(encoded.as_bytes()).unwrap(),
629        mailbox
630    );
631}
632
633#[cfg(test)]
634#[test]
635fn test_crlf_injection_via_display_name() {
636    // A display name may pick up a stray CR/LF, for example a value imported
637    // from another system with an embedded line break. Rewriting it to a space
638    // keeps it from terminating the header line: re-parsing the header block
639    // yields a single From header, not a spurious second one.
640    let mailbox = Mailbox {
641        name: Some("Ada Lovelace\r\nNotes: imported".to_string()),
642        address: AddrSpec::new("alex", "example.com"),
643    };
644    let encoded = mailbox.encode_value().to_string();
645    k9::snapshot!(
646        BString::from(encoded.clone()),
647        r#""Ada Lovelace  Notes: imported" <alex@example.com>"#
648    );
649
650    let header_block = format!("From: {encoded}\r\n\r\n");
651    let parsed = crate::Header::parse_headers(header_block).unwrap();
652    let names: Vec<String> = parsed
653        .headers
654        .iter()
655        .map(|h| h.get_name().to_string())
656        .collect();
657    k9::snapshot!(
658        names,
659        r#"
660[
661    "From",
662]
663"#
664    );
665}
666
667#[cfg(test)]
668#[test]
669fn test_bare_lf_injection_via_display_name() {
670    // Same as test_crlf_injection_via_display_name, for a bare LF with no
671    // preceding CR: quote_string's fold check has a separate match arm for
672    // this case, so it needs its own regression coverage.
673    let mailbox = Mailbox {
674        name: Some("Ada Lovelace\nNotes: imported".to_string()),
675        address: AddrSpec::new("alex", "example.com"),
676    };
677    let encoded = mailbox.encode_value().to_string();
678    k9::snapshot!(
679        BString::from(encoded.clone()),
680        r#""Ada Lovelace Notes: imported" <alex@example.com>"#
681    );
682
683    let header_block = format!("From: {encoded}\r\n\r\n");
684    let parsed = crate::Header::parse_headers(header_block).unwrap();
685    let names: Vec<String> = parsed
686        .headers
687        .iter()
688        .map(|h| h.get_name().to_string())
689        .collect();
690    k9::snapshot!(
691        names,
692        r#"
693[
694    "From",
695]
696"#
697    );
698}
699
700#[cfg(test)]
701#[test]
702fn test_encode_folds_non_ascii_mailbox_display_name() {
703    // Use a non-ASCII name long enough that qp_encode itself folds it into
704    // multiple encoded-words joined by "\r\n\t". The addr-spec fold decision
705    // must measure only the last physical line of the encoded phrase, not the
706    // total length of the phrase, or it would spuriously trigger another fold
707    // before <addr> regardless of how short the last line actually is. No
708    // round-trip assertion: decoding a qp_encode fold that lands mid-word
709    // reconstructs a space at the boundary, a separate, pre-existing lossy
710    // round-trip in the phrase parser.
711    let mailbox = Mailbox {
712        name: Some("日本語の非常に長い表示名前です本当に長いですよ".repeat(3)),
713        address: AddrSpec::new("alex", "example.com"),
714    };
715    let encoded = mailbox.encode_value().to_string();
716    // Whether to insert a fold before <addr> is decided by whether appending
717    // <addr> to the last qp_encode line would exceed the fold width, using the
718    // length of that last physical line rather than the total encoded length.
719    // Here it does not exceed the width, so no further fold is inserted.
720    k9::snapshot!(
721        BString::from(encoded.clone()),
722        r#"
723=?UTF-8?q?=E6=97=A5=E6=9C=AC=E8=AA=9E=E3=81=AE=E9=9D=9E=E5=B8=B8?=\r
724\t=?UTF-8?q?=E3=81=AB=E9=95=B7=E3=81=84=E8=A1=A8=E7=A4=BA=E5=90=8D?=\r
725\t=?UTF-8?q?=E5=89=8D=E3=81=A7=E3=81=99=E6=9C=AC=E5=BD=93=E3=81=AB?=\r
726\t=?UTF-8?q?=E9=95=B7=E3=81=84=E3=81=A7=E3=81=99=E3=82=88=E6=97=A5?=\r
727\t=?UTF-8?q?=E6=9C=AC=E8=AA=9E=E3=81=AE=E9=9D=9E=E5=B8=B8=E3=81=AB?=\r
728\t=?UTF-8?q?=E9=95=B7=E3=81=84=E8=A1=A8=E7=A4=BA=E5=90=8D=E5=89=8D?=\r
729\t=?UTF-8?q?=E3=81=A7=E3=81=99=E6=9C=AC=E5=BD=93=E3=81=AB=E9=95=B7?=\r
730\t=?UTF-8?q?=E3=81=84=E3=81=A7=E3=81=99=E3=82=88=E6=97=A5=E6=9C=AC?=\r
731\t=?UTF-8?q?=E8=AA=9E=E3=81=AE=E9=9D=9E=E5=B8=B8=E3=81=AB=E9=95=B7?=\r
732\t=?UTF-8?q?=E3=81=84=E8=A1=A8=E7=A4=BA=E5=90=8D=E5=89=8D=E3=81=A7?=\r
733\t=?UTF-8?q?=E3=81=99=E6=9C=AC=E5=BD=93=E3=81=AB=E9=95=B7=E3=81=84?=\r
734\t=?UTF-8?q?=E3=81=A7=E3=81=99=E3=82=88?= <alex@example.com>
735"#
736    );
737    Parser::parse_mailbox_header(encoded.as_bytes()).unwrap();
738}
739
740#[cfg(test)]
741#[test]
742fn test_encode_folds_mailbox_list_at_boundaries() {
743    let list = MailboxList(vec![
744        Mailbox {
745            name: Some(
746                "The Honorable Regional Manager of the Northwestern Sales Territory".to_string(),
747            ),
748            address: AddrSpec::new("alex", "example.com"),
749        },
750        Mailbox {
751            name: Some("Bob Smith".to_string()),
752            address: AddrSpec::new("bob", "example.com"),
753        },
754    ]);
755    let encoded = list.encode_value().to_string();
756    k9::snapshot!(
757        BString::from(encoded.clone()),
758        r#"
759"The Honorable Regional Manager of the Northwestern Sales Territory"\r
760\t<alex@example.com>,\r
761\t"Bob Smith" <bob@example.com>
762"#
763    );
764    k9::assert_equal!(
765        Parser::parse_mailbox_list_header(encoded.as_bytes()).unwrap(),
766        list
767    );
768}
769
770#[cfg(test)]
771#[test]
772fn test_encode_folds_address_list_at_boundaries() {
773    // Same as the mailbox-list case, for the address-list headers
774    // (To/Cc/Bcc/Reply-To).
775    let list = AddressList(vec![
776        Address::Mailbox(Mailbox {
777            name: Some(
778                "The Honorable Regional Manager of the Northwestern Sales Territory".to_string(),
779            ),
780            address: AddrSpec::new("alex", "example.com"),
781        }),
782        Address::Mailbox(Mailbox {
783            name: Some("Bob Smith".to_string()),
784            address: AddrSpec::new("bob", "example.com"),
785        }),
786    ]);
787    let encoded = list.encode_value().to_string();
788    k9::snapshot!(
789        BString::from(encoded.clone()),
790        r#"
791"The Honorable Regional Manager of the Northwestern Sales Territory"\r
792\t<alex@example.com>,\r
793\t"Bob Smith" <bob@example.com>
794"#
795    );
796    k9::assert_equal!(
797        Parser::parse_address_list_header(encoded.as_bytes()).unwrap(),
798        list
799    );
800}
801
802#[cfg(test)]
803#[test]
804fn test_obs_local_part_with_special_chars() {
805    // obs-local-part where the quoted-string word contains characters
806    // that require quoting (space, specials)
807    k9::snapshot!(
808        parse_with(r#""hello world".user@example.com"#.as_bytes(), addr_spec),
809        r#"
810Ok(
811    AddrSpec {
812        local_part: "hello world.user",
813        domain: "example.com",
814    },
815)
816"#
817    );
818    // Verify the round-trip encodes as a valid RFC 5321 quoted-string
819    let addr = AddrSpec::new("hello world.user", "example.com");
820    k9::assert_equal!(addr.encode_value(), r#""hello world.user"@example.com"#);
821}
822
823#[cfg(test)]
824#[test]
825fn test_utf8_non_ascii_in_local_part() {
826    // RFC 6531/6532: internationalized local-part with non-ASCII characters
827    k9::snapshot!(
828        parse_with("用户@example.com".as_bytes(), addr_spec),
829        r#"
830Ok(
831    AddrSpec {
832        local_part: "用户",
833        domain: "example.com",
834    },
835)
836"#
837    );
838    k9::snapshot!(
839        parse_with("münchen@example.com".as_bytes(), addr_spec),
840        r#"
841Ok(
842    AddrSpec {
843        local_part: "münchen",
844        domain: "example.com",
845    },
846)
847"#
848    );
849}
850
851#[cfg(test)]
852#[test]
853fn test_utf8_non_ascii_in_domain() {
854    // RFC 6531: internationalized domain in header address
855    k9::snapshot!(
856        parse_with("user@例え.jp".as_bytes(), addr_spec),
857        r#"
858Ok(
859    AddrSpec {
860        local_part: "user",
861        domain: "例え.jp",
862    },
863)
864"#
865    );
866}
867
868#[cfg(test)]
869#[test]
870fn test_quoted_pair_non_ascii() {
871    // quoted_pair with utf8_non_ascii: backslash followed by a non-ASCII char
872    k9::snapshot!(
873        parse_with(r#""\München"@example.com"#.as_bytes(), addr_spec),
874        r#"
875Ok(
876    AddrSpec {
877        local_part: "München",
878        domain: "example.com",
879    },
880)
881"#
882    );
883}
884
885#[cfg(test)]
886#[test]
887fn test_invalid_utf8_rejected() {
888    // Lone continuation byte (0x80) is not valid UTF-8 and should be rejected
889    // in atext position
890    let input = b"user\x80@example.com";
891    parse_with(input, addr_spec).unwrap_err();
892
893    // Overlong encoding of '/' (U+002F): 0xC0 0xAF is invalid UTF-8
894    let input = b"user\xC0\xAF@example.com";
895    parse_with(input, addr_spec).unwrap_err();
896
897    // Truncated multi-byte sequence: 0xC3 without continuation
898    let input = b"user\xC3@example.com";
899    parse_with(input, addr_spec).unwrap_err();
900
901    // Invalid byte in quoted-string qtext position
902    let input = b"\"user\x80\"@example.com";
903    parse_with(input, addr_spec).unwrap_err();
904
905    // Invalid byte in comment ctext position
906    let input = b"(comment\x80) user@example.com";
907    parse_with(input, mailbox).unwrap_err();
908}
909
910// atom = { cfws? ~ atext ~ cfws? }
911fn atom(input: Span) -> IResult<Span, BString> {
912    let (loc, text) = context("atom", delimited(opt(cfws), atext, opt(cfws))).parse(input)?;
913    Ok((loc, (*text).into()))
914}
915
916// word = { atom | quoted_string }
917fn word(input: Span) -> IResult<Span, BString> {
918    context("word", alt((atom, quoted_string))).parse(input)
919}
920
921// obs_local_part = { word ~ (dot ~ word)* }
922fn obs_local_part(input: Span) -> IResult<Span, BString> {
923    let (loc, (word, dotted_words)) =
924        context("obs_local_part", (word, many0((tag("."), word)))).parse(input)?;
925    let mut result = word;
926
927    for (_dot, w) in dotted_words {
928        result.push(b'.');
929        result.push_str(&w);
930    }
931
932    Ok((loc, result))
933}
934
935// local_part = { dot_atom | quoted_string | obs_local_part }
936// obs_local_part (word *("." word)) is a superset of both dot_atom and
937// quoted_string: a dot-separated run of atoms is an obs_local_part where
938// every word is an atom, and a bare quoted-string is an obs_local_part
939// with no dot continuations. It must be tried first because dot_atom can
940// partially match (e.g. consuming "first" from "first.\"last\"@domain")
941// and then fail in the wider addr_spec context with no backtracking.
942fn local_part(input: Span) -> IResult<Span, BString> {
943    context("local_part", alt((obs_local_part, dot_atom, quoted_string))).parse(input)
944}
945
946// domain = { dot_atom | domain_literal | obs_domain }
947fn domain(input: Span) -> IResult<Span, BString> {
948    context("domain", alt((dot_atom, domain_literal, obs_domain))).parse(input)
949}
950
951// obs_domain = { atom ~ ( dot ~ atom)* }
952fn obs_domain(input: Span) -> IResult<Span, BString> {
953    let (loc, (atom, dotted_atoms)) =
954        context("obs_domain", (atom, many0((tag("."), atom)))).parse(input)?;
955    let mut result = atom;
956
957    for (_dot, w) in dotted_atoms {
958        result.push(b'.');
959        result.push_str(&w);
960    }
961
962    Ok((loc, result))
963}
964
965// domain_literal = { cfws? ~ "[" ~ (fws? ~ dtext)* ~ fws? ~ "]" ~ cfws? }
966fn domain_literal(input: Span) -> IResult<Span, BString> {
967    let (loc, (bits, trailer)) = context(
968        "domain_literal",
969        delimited(
970            opt(cfws),
971            delimited(
972                tag("["),
973                (
974                    many0((
975                        opt(fws),
976                        alt((
977                            take_while_m_n(1, 1, is_dtext_ascii),
978                            utf8_non_ascii,
979                            quoted_pair,
980                        )),
981                    )),
982                    opt(fws),
983                ),
984                tag("]"),
985            ),
986            opt(cfws),
987        ),
988    )
989    .parse(input)?;
990
991    let mut result = BString::default();
992    result.push(b'[');
993    for (a, b) in bits {
994        if let Some(a) = a {
995            result.push_str(a);
996        }
997        result.push_str(b);
998    }
999    if let Some(t) = trailer {
1000        result.push_str(t);
1001    }
1002    result.push(b']');
1003    Ok((loc, result))
1004}
1005
1006// dot_atom_text = @{ atext ~ ("." ~ atext)* }
1007fn dot_atom_text(input: Span) -> IResult<Span, BString> {
1008    let (loc, (a, b)) =
1009        context("dot_atom_text", (atext, many0(preceded(tag("."), atext)))).parse(input)?;
1010    let mut result: BString = (*a).into();
1011    for item in b {
1012        result.push(b'.');
1013        result.push_str(item);
1014    }
1015
1016    Ok((loc, result))
1017}
1018
1019// dot_atom = { cfws? ~ dot_atom_text ~ cfws? }
1020fn dot_atom(input: Span) -> IResult<Span, BString> {
1021    context("dot_atom", delimited(opt(cfws), dot_atom_text, opt(cfws))).parse(input)
1022}
1023
1024#[cfg(test)]
1025#[test]
1026fn test_dot_atom() {
1027    k9::snapshot!(
1028        parse_with("hello".as_bytes(), dot_atom),
1029        r#"
1030Ok(
1031    "hello",
1032)
1033"#
1034    );
1035
1036    k9::snapshot!(
1037        parse_with("hello.there".as_bytes(), dot_atom),
1038        r#"
1039Ok(
1040    "hello.there",
1041)
1042"#
1043    );
1044
1045    k9::snapshot!(
1046        parse_with("hello.".as_bytes(), dot_atom),
1047        r#"
1048Err(
1049    HeaderParse(
1050        "Error at line 1, in Eof:
1051hello.
1052     ^
1053
1054",
1055    ),
1056)
1057"#
1058    );
1059
1060    k9::snapshot!(
1061        parse_with("(wat)hello".as_bytes(), dot_atom),
1062        r#"
1063Ok(
1064    "hello",
1065)
1066"#
1067    );
1068}
1069
1070// cfws = { ( (fws? ~ comment)+ ~ fws?) | fws }
1071fn cfws(input: Span) -> IResult<Span, Span> {
1072    context(
1073        "cfws",
1074        recognize(alt((
1075            recognize((many1((opt(fws), comment)), opt(fws))),
1076            fws,
1077        ))),
1078    )
1079    .parse(input)
1080}
1081
1082// comment = { "(" ~ (fws? ~ (ccontent_atom | comment))* ~ fws? ~ ")" }
1083fn comment(input: Span) -> IResult<Span, Span> {
1084    // Track nesting depth explicitly instead of recursing to prevent a deeply
1085    // nested comment from exhausting the stack.
1086    context(
1087        "comment",
1088        recognize(|input| {
1089            let (mut input, _) = tag("(").parse(input)?;
1090            let mut depth = 1usize;
1091
1092            while depth > 0 {
1093                let (remaining, _) = opt(fws).parse(input)?;
1094                input = remaining;
1095
1096                match input.fragment().first() {
1097                    Some(b'(') => {
1098                        (input, _) = tag("(").parse(input)?;
1099                        depth += 1;
1100                    }
1101                    Some(b')') => {
1102                        (input, _) = tag(")").parse(input)?;
1103                        depth -= 1;
1104                    }
1105                    _ => {
1106                        (input, _) = ccontent_atom.parse(input)?;
1107                    }
1108                }
1109            }
1110
1111            Ok((input, ()))
1112        }),
1113    )
1114    .parse(input)
1115}
1116
1117#[cfg(test)]
1118#[test]
1119fn test_comment() {
1120    k9::snapshot!(
1121        BStr::new(&parse_with("(wat)".as_bytes(), comment).unwrap()),
1122        "(wat)"
1123    );
1124}
1125
1126#[cfg(test)]
1127#[test]
1128fn deeply_nested_comment_does_not_overflow_the_stack() {
1129    let input = format!(
1130        "probe@example.invalid {}{}",
1131        "(".repeat(10_000),
1132        ")".repeat(10_000)
1133    );
1134
1135    k9::assert_equal!(
1136        parse_with(input.as_bytes(), mailbox).unwrap(),
1137        Mailbox {
1138            name: None,
1139            address: AddrSpec {
1140                local_part: "probe".to_string(),
1141                domain: "example.invalid".to_string(),
1142            },
1143        }
1144    );
1145}
1146
1147// ccontent = { ctext | quoted_pair | encoded_word }
1148fn ccontent_atom(input: Span) -> IResult<Span, Span> {
1149    context(
1150        "ccontent_atom",
1151        recognize(alt((
1152            recognize(alt((take_while_m_n(1, 1, is_ctext_ascii), utf8_non_ascii))),
1153            recognize(quoted_pair),
1154            recognize(encoded_word),
1155        ))),
1156    )
1157    .parse(input)
1158}
1159
1160/// Remove CFWS (comments and folding whitespace) from a header value, returning
1161/// the surviving tokens joined by single spaces, or `None` if it does not
1162/// tokenize cleanly.
1163pub(crate) fn strip_cfws(input: &str) -> Option<String> {
1164    fn token(input: Span) -> IResult<Span, Span> {
1165        take_while1(|c| !matches!(c, b' ' | b'\t' | b'\r' | b'\n' | b'(')).parse(input)
1166    }
1167    fn tokens(input: Span) -> IResult<Span, Vec<Option<Span>>> {
1168        many0(alt((map(cfws, |_| None), map(token, Some)))).parse(input)
1169    }
1170
1171    let mut out: Vec<u8> = Vec::new();
1172    for token in parse_with(input.as_bytes(), tokens)
1173        .ok()?
1174        .into_iter()
1175        .flatten()
1176    {
1177        if !out.is_empty() {
1178            out.push(b' ');
1179        }
1180        out.extend_from_slice(token.fragment());
1181    }
1182    String::from_utf8(out).ok()
1183}
1184
1185#[cfg(test)]
1186#[test]
1187fn test_strip_cfws() {
1188    k9::assert_equal!(
1189        strip_cfws("Thu, 02 Jul 26 18:55:38 UTC (Coordinated)").unwrap(),
1190        "Thu, 02 Jul 26 18:55:38 UTC"
1191    );
1192    k9::assert_equal!(
1193        strip_cfws("Thu, 02 Jul 26 (comment) 18:55:38 UTC").unwrap(),
1194        "Thu, 02 Jul 26 18:55:38 UTC"
1195    );
1196    // Nested comments and quoted parens are handled by the comment parser.
1197    k9::assert_equal!(strip_cfws("a (b (c) \\) d) e").unwrap(), "a e");
1198}
1199
1200fn is_quoted_pair_ascii(c: u8) -> bool {
1201    match c {
1202        0x00 | b'\r' | b'\n' | b' ' => true,
1203        c => is_obs_no_ws_ctl(c) || is_vchar_ascii(c),
1204    }
1205}
1206
1207/// Byte-level predicate for quoted_pair, including UTF-8 continuation/leading
1208/// bytes. Used for non-parser checks. For parsing, use `quoted_pair` which
1209/// validates UTF-8 via `utf8_non_ascii`.
1210fn is_quoted_pair(c: u8) -> bool {
1211    is_quoted_pair_ascii(c) || c >= 0x80
1212}
1213
1214// quoted_pair = { ( "\\"  ~ (vchar | wsp)) | obs_qp }
1215// obs_qp = { "\\" ~ ( "\u{00}" | obs_no_ws_ctl | "\r" | "\n") }
1216fn quoted_pair(input: Span) -> IResult<Span, Span> {
1217    context(
1218        "quoted_pair",
1219        preceded(
1220            tag("\\"),
1221            alt((take_while_m_n(1, 1, is_quoted_pair_ascii), utf8_non_ascii)),
1222        ),
1223    )
1224    .parse(input)
1225}
1226
1227// encoded_word = { "=?" ~ charset ~ ("*" ~ language)? ~ "?" ~ encoding ~ "?" ~ encoded_text ~ "?=" }
1228fn encoded_word(input: Span) -> IResult<Span, BString> {
1229    let (loc, (charset, _language, _, encoding, _, text)) = context(
1230        "encoded_word",
1231        delimited(
1232            tag("=?"),
1233            (
1234                charset,
1235                opt(preceded(tag("*"), language)),
1236                tag("?"),
1237                encoding,
1238                tag("?"),
1239                encoded_text,
1240            ),
1241            tag("?="),
1242        ),
1243    )
1244    .parse(input)?;
1245
1246    let bytes = match *encoding.fragment() {
1247        b"B" | b"b" => data_encoding::BASE64_MIME
1248            .decode(text.as_bytes())
1249            .map_err(|err| {
1250                make_context_error(
1251                    input,
1252                    format!("encoded_word: base64 decode failed: {err:#}"),
1253                )
1254            })?,
1255        b"Q" | b"q" => {
1256            // for rfc2047 header encoding, _ can be used to represent a space
1257            let munged = text.replace("_", " ");
1258            // The quoted_printable crate will unhelpfully strip trailing space
1259            // from the decoded input string, and we must track and restore it
1260            let had_trailing_space = munged.ends_with_str(" ");
1261            let mut decoded = quoted_printable::decode(munged, quoted_printable::ParseMode::Robust)
1262                .map_err(|err| {
1263                    make_context_error(
1264                        input,
1265                        format!("encoded_word: quoted printable decode failed: {err:#}"),
1266                    )
1267                })?;
1268            if had_trailing_space && !decoded.ends_with(b" ") {
1269                decoded.push(b' ');
1270            }
1271            decoded
1272        }
1273        encoding => {
1274            let encoding = BStr::new(encoding);
1275            return Err(make_context_error(
1276                input,
1277                format!(
1278                    "encoded_word: invalid encoding '{encoding}', expected one of b, B, q or Q"
1279                ),
1280            ));
1281        }
1282    };
1283
1284    let charset_name = charset.to_str().map_err(|err| {
1285        make_context_error(
1286            input,
1287            format!(
1288                "encoded_word: charset {} is not UTF-8: {err}",
1289                BStr::new(*charset)
1290            ),
1291        )
1292    })?;
1293
1294    let charset = Encoding::by_name(&*charset_name).ok_or_else(|| {
1295        make_context_error(
1296            input,
1297            format!("encoded_word: unsupported charset '{charset_name}'"),
1298        )
1299    })?;
1300
1301    let decoded = charset.decode_simple(&bytes).map_err(|err| {
1302        make_context_error(
1303            input,
1304            format!("encoded_word: failed to decode as '{charset_name}': {err}"),
1305        )
1306    })?;
1307
1308    Ok((loc, decoded.into()))
1309}
1310
1311// charset = @{ (!"*" ~ token)+ }
1312fn charset(input: Span) -> IResult<Span, Span> {
1313    context("charset", take_while1(|c| c != b'*' && is_token(c))).parse(input)
1314}
1315
1316// language = @{ token+ }
1317fn language(input: Span) -> IResult<Span, Span> {
1318    context("language", take_while1(|c| c != b'*' && is_token(c))).parse(input)
1319}
1320
1321// encoding = @{ token+ }
1322fn encoding(input: Span) -> IResult<Span, Span> {
1323    context("encoding", take_while1(|c| c != b'*' && is_token(c))).parse(input)
1324}
1325
1326// encoded_text = @{ (!( " " | "?") ~ vchar)+ }
1327fn encoded_text(input: Span) -> IResult<Span, Span> {
1328    context(
1329        "encoded_text",
1330        recognize(many1(alt((
1331            take_while1(|c| is_vchar_ascii(c) && c != b' ' && c != b'?'),
1332            utf8_non_ascii,
1333        )))),
1334    )
1335    .parse(input)
1336}
1337
1338// quoted_string = { cfws? ~ "\"" ~ (fws? ~ qcontent)* ~ fws? ~ "\"" ~ cfws? }
1339fn quoted_string(input: Span) -> IResult<Span, BString> {
1340    let (loc, (bits, trailer)) = context(
1341        "quoted_string",
1342        delimited(
1343            opt(cfws),
1344            delimited(
1345                tag("\""),
1346                (many0((opt(fws), qcontent)), opt(fws)),
1347                tag("\""),
1348            ),
1349            opt(cfws),
1350        ),
1351    )
1352    .parse(input)?;
1353
1354    let mut result = BString::default();
1355    for (a, b) in bits {
1356        if let Some(a) = a {
1357            result.push_str(a);
1358        }
1359        result.push_str(b);
1360    }
1361    if let Some(t) = trailer {
1362        result.push_str(t);
1363    }
1364    Ok((loc, result))
1365}
1366
1367// qcontent = { qtext | quoted_pair }
1368fn qcontent(input: Span) -> IResult<Span, Span> {
1369    context(
1370        "qcontent",
1371        alt((
1372            take_while_m_n(1, 1, is_qtext_ascii),
1373            utf8_non_ascii,
1374            quoted_pair,
1375        )),
1376    )
1377    .parse(input)
1378}
1379
1380fn content_id(input: Span) -> IResult<Span, MessageID> {
1381    let (loc, id) = context("content_id", msg_id).parse(input)?;
1382    Ok((loc, id))
1383}
1384
1385fn msg_id(input: Span) -> IResult<Span, MessageID> {
1386    let (loc, id) = context("msg_id", alt((strict_msg_id, relaxed_msg_id))).parse(input)?;
1387    Ok((loc, id))
1388}
1389
1390fn relaxed_msg_id(input: Span) -> IResult<Span, MessageID> {
1391    let (loc, id) = context(
1392        "msg_id",
1393        delimited(
1394            preceded(opt(cfws), tag("<")),
1395            many0(take_while_m_n(1, 1, not_angle)),
1396            preceded(tag(">"), opt(cfws)),
1397        ),
1398    )
1399    .parse(input)?;
1400
1401    let mut result = BString::default();
1402    for item in id.into_iter() {
1403        result.push_str(*item);
1404    }
1405
1406    Ok((loc, MessageID(result)))
1407}
1408
1409// msg_id_list = { msg_id+ }
1410fn msg_id_list(input: Span) -> IResult<Span, Vec<MessageID>> {
1411    context("msg_id_list", many1(msg_id)).parse(input)
1412}
1413
1414// id_left = { dot_atom_text | obs_id_left }
1415// obs_id_left = { local_part }
1416fn id_left(input: Span) -> IResult<Span, BString> {
1417    context("id_left", alt((dot_atom_text, local_part))).parse(input)
1418}
1419
1420// id_right = { dot_atom_text | no_fold_literal | obs_id_right }
1421// obs_id_right = { domain }
1422fn id_right(input: Span) -> IResult<Span, BString> {
1423    context("id_right", alt((dot_atom_text, no_fold_literal, domain))).parse(input)
1424}
1425
1426// no_fold_literal = { "[" ~ dtext* ~ "]" }
1427fn no_fold_literal(input: Span) -> IResult<Span, BString> {
1428    context(
1429        "no_fold_literal",
1430        map(
1431            recognize((
1432                tag("["),
1433                recognize(many0(alt((take_while1(is_dtext_ascii), utf8_non_ascii)))),
1434                tag("]"),
1435            )),
1436            |s: Span| (*s).into(),
1437        ),
1438    )
1439    .parse(input)
1440}
1441
1442// msg_id = { cfws? ~ "<" ~ id_left ~ "@" ~ id_right ~ ">" ~ cfws? }
1443fn strict_msg_id(input: Span) -> IResult<Span, MessageID> {
1444    let (loc, (left, _, right)) = context(
1445        "msg_id",
1446        delimited(
1447            preceded(opt(cfws), tag("<")),
1448            (id_left, tag("@"), id_right),
1449            preceded(tag(">"), opt(cfws)),
1450        ),
1451    )
1452    .parse(input)?;
1453
1454    let mut result: BString = left;
1455    result.push_char('@');
1456    result.push_str(right);
1457
1458    Ok((loc, MessageID(result)))
1459}
1460
1461// obs_unstruct = { (( "\r"* ~ "\n"* ~ ((encoded_word | obs_utext)~ "\r"* ~ "\n"*)+) | fws)+ }
1462fn unstructured(input: Span) -> IResult<Span, BString> {
1463    #[derive(Debug)]
1464    enum Word {
1465        Encoded(BString),
1466        UText(BString),
1467        Fws,
1468    }
1469
1470    let (loc, words) = context(
1471        "unstructured",
1472        many0(alt((
1473            preceded(
1474                map(take_while(|c| c == b'\r' || c == b'\n'), |_| Word::Fws),
1475                terminated(
1476                    alt((
1477                        map(encoded_word, Word::Encoded),
1478                        map(obs_utext, |s| Word::UText((*s).into())),
1479                    )),
1480                    map(take_while(|c| c == b'\r' || c == b'\n'), |_| Word::Fws),
1481                ),
1482            ),
1483            map(fws, |_| Word::Fws),
1484        ))),
1485    )
1486    .parse(input)?;
1487
1488    #[derive(Debug)]
1489    enum ProcessedWord {
1490        Encoded(BString),
1491        Text(BString),
1492        Fws,
1493    }
1494    let mut processed = vec![];
1495    for w in words {
1496        match w {
1497            Word::Encoded(p) => {
1498                if processed.len() >= 2
1499                    && matches!(processed.last(), Some(ProcessedWord::Fws))
1500                    && matches!(processed[processed.len() - 2], ProcessedWord::Encoded(_))
1501                {
1502                    // Fws between encoded words is elided
1503                    processed.pop();
1504                }
1505                processed.push(ProcessedWord::Encoded(p));
1506            }
1507            Word::Fws => {
1508                // Collapse runs of Fws/newline to a single Fws
1509                if !matches!(processed.last(), Some(ProcessedWord::Fws)) {
1510                    processed.push(ProcessedWord::Fws);
1511                }
1512            }
1513            Word::UText(c) => match processed.last_mut() {
1514                Some(ProcessedWord::Text(prior)) => prior.push_str(c),
1515                _ => processed.push(ProcessedWord::Text(c)),
1516            },
1517        }
1518    }
1519
1520    let mut result = BString::default();
1521    for word in processed {
1522        match word {
1523            ProcessedWord::Encoded(s) | ProcessedWord::Text(s) => {
1524                result.push_str(&s);
1525            }
1526            ProcessedWord::Fws => {
1527                result.push(b' ');
1528            }
1529        }
1530    }
1531
1532    Ok((loc, result))
1533}
1534
1535fn arc_authentication_results(input: Span) -> IResult<Span, ARCAuthenticationResults> {
1536    context(
1537        "arc_authentication_results",
1538        map(
1539            (
1540                preceded(opt(cfws), tag("i")),
1541                preceded(opt(cfws), tag("=")),
1542                preceded(opt(cfws), nom::character::complete::u8),
1543                preceded(opt(cfws), tag(";")),
1544                preceded(opt(cfws), value),
1545                opt(preceded(cfws, nom::character::complete::u32)),
1546                alt((no_result, many1(resinfo))),
1547                opt(cfws),
1548            ),
1549            |(_i, _eq, instance, _semic, serv_id, version, results, _)| ARCAuthenticationResults {
1550                instance,
1551                serv_id,
1552                version,
1553                results,
1554            },
1555        ),
1556    )
1557    .parse(input)
1558}
1559
1560fn authentication_results(input: Span) -> IResult<Span, AuthenticationResults> {
1561    context(
1562        "authentication_results",
1563        map(
1564            (
1565                preceded(opt(cfws), value),
1566                opt(preceded(cfws, nom::character::complete::u32)),
1567                alt((no_result, many1(resinfo))),
1568                opt(cfws),
1569            ),
1570            |(serv_id, version, results, _)| AuthenticationResults {
1571                serv_id,
1572                version,
1573                results,
1574            },
1575        ),
1576    )
1577    .parse(input)
1578}
1579
1580fn no_result(input: Span) -> IResult<Span, Vec<AuthenticationResult>> {
1581    context(
1582        "no_result",
1583        map((opt(cfws), tag(";"), opt(cfws), tag("none")), |_| vec![]),
1584    )
1585    .parse(input)
1586}
1587
1588fn resinfo(input: Span) -> IResult<Span, AuthenticationResult> {
1589    context(
1590        "resinfo",
1591        map(
1592            (
1593                opt(cfws),
1594                tag(";"),
1595                methodspec,
1596                opt(preceded(cfws, reasonspec)),
1597                opt(many1(propspec)),
1598            ),
1599            |(_, _, (method, method_version, result), reason, props)| AuthenticationResult {
1600                method,
1601                method_version,
1602                result,
1603                reason,
1604                props: match props {
1605                    None => BTreeMap::default(),
1606                    Some(props) => props.into_iter().collect(),
1607                },
1608            },
1609        ),
1610    )
1611    .parse(input)
1612}
1613
1614fn methodspec(input: Span) -> IResult<Span, (String, Option<u32>, String)> {
1615    context(
1616        "methodspec",
1617        map(
1618            (
1619                opt(cfws),
1620                (keyword, opt(methodversion)),
1621                opt(cfws),
1622                tag("="),
1623                opt(cfws),
1624                keyword,
1625            ),
1626            |(_, (method, methodversion), _, _, _, result)| (method, methodversion, result),
1627        ),
1628    )
1629    .parse(input)
1630}
1631
1632// Taken from https://datatracker.ietf.org/doc/html/rfc8601 which says
1633// that this is the same as the SMTP Keyword token (RFC 5321 section 4.1.2).
1634// Keyword = Ldh-str = *( ALPHA / DIGIT / "-" ) Let-dig
1635// Only matches ASCII alphanumeric and '-'.
1636fn keyword(input: Span) -> IResult<Span, String> {
1637    context(
1638        "keyword",
1639        map(
1640            take_while1(|c: u8| c.is_ascii_alphanumeric() || c == b'-'),
1641            // SAFETY: predicate only matches ASCII bytes
1642            |s: Span| String::from_utf8((*s).into()).expect("keyword is ASCII-only"),
1643        ),
1644    )
1645    .parse(input)
1646}
1647
1648fn methodversion(input: Span) -> IResult<Span, u32> {
1649    context(
1650        "methodversion",
1651        preceded(
1652            (opt(cfws), tag("/"), opt(cfws)),
1653            nom::character::complete::u32,
1654        ),
1655    )
1656    .parse(input)
1657}
1658
1659fn reasonspec(input: Span) -> IResult<Span, BString> {
1660    context(
1661        "reason",
1662        map(
1663            (tag("reason"), opt(cfws), tag("="), opt(cfws), value),
1664            |(_, _, _, _, value)| value,
1665        ),
1666    )
1667    .parse(input)
1668}
1669
1670fn propspec(input: Span) -> IResult<Span, (String, BString)> {
1671    context(
1672        "propspec",
1673        map(
1674            (
1675                // RFC 8601 resinfo ABNF says CFWS is required before each
1676                // propspec, but we use opt(cfws) here because other parsers
1677                // (notably quoted_string) may have already consumed the
1678                // whitespace.
1679                opt(cfws),
1680                keyword,
1681                opt(cfws),
1682                tag("."),
1683                opt(cfws),
1684                keyword,
1685                opt(cfws),
1686                tag("="),
1687                opt(cfws),
1688                // pvalue = [CFWS] ( value / [ [CFWS] "@" ] domain ) [CFWS]
1689                // Try @domain and local@domain first (distinctive @ marker),
1690                // then quoted_string (distinctive " marker), then domain
1691                // (handles dotted names), then mime_token last (single tokens).
1692                alt((
1693                    map(preceded(tag("@"), domain), |d| {
1694                        let mut at_dom = BString::from("@");
1695                        at_dom.push_str(d);
1696                        at_dom
1697                    }),
1698                    map(separated_pair(local_part, tag("@"), domain), |(u, d)| {
1699                        let mut result: BString = u;
1700                        result.push(b'@');
1701                        result.push_str(d);
1702                        result
1703                    }),
1704                    quoted_string,
1705                    domain,
1706                    map(mime_token, |s: Span| (*s).into()),
1707                )),
1708                opt(cfws),
1709            ),
1710            |(_, ptype, _, _, _, property, _, _, _, value, _)| {
1711                (format!("{ptype}.{property}"), value)
1712            },
1713        ),
1714    )
1715    .parse(input)
1716}
1717
1718// obs_utext = @{ "\u{00}" | obs_no_ws_ctl | vchar }
1719fn obs_utext(input: Span) -> IResult<Span, Span> {
1720    context(
1721        "obs_utext",
1722        alt((
1723            take_while_m_n(1, 1, |c| {
1724                c == 0x00 || is_obs_no_ws_ctl(c) || is_vchar_ascii(c)
1725            }),
1726            utf8_non_ascii,
1727        )),
1728    )
1729    .parse(input)
1730}
1731
1732fn is_mime_token(c: u8) -> bool {
1733    is_char(c) && c != b' ' && !is_ctl(c) && !is_tspecial(c)
1734}
1735
1736// mime_token = { (!(" " | ctl | tspecials) ~ char)+ }
1737// Also accepts validated UTF-8 multi-byte sequences per RFC 6532.
1738fn mime_token(input: Span) -> IResult<Span, Span> {
1739    context(
1740        "mime_token",
1741        recognize(many1(alt((take_while1(is_mime_token), utf8_non_ascii)))),
1742    )
1743    .parse(input)
1744}
1745
1746// RFC2045 modified by RFC2231 MIME header fields
1747// content_type = { cfws? ~ mime_type ~ cfws? ~ "/" ~ cfws? ~ subtype ~
1748//  cfws? ~ (";"? ~ cfws? ~ parameter ~ cfws?)*
1749// }
1750fn content_type(input: Span) -> IResult<Span, MimeParameters> {
1751    let (loc, (mime_type, _, _, _, mime_subtype, _, parameters)) = context(
1752        "content_type",
1753        preceded(
1754            opt(cfws),
1755            (
1756                mime_token,
1757                opt(cfws),
1758                tag("/"),
1759                opt(cfws),
1760                mime_token,
1761                opt(cfws),
1762                many0(preceded(
1763                    // Note that RFC 2231 is a bit of a mess, showing examples
1764                    // without `;` as a separator in the original text, but
1765                    // in the errata from several years later, corrects those
1766                    // to show the `;`.
1767                    // In the meantime, there are implementations that assume
1768                    // that the `;` is optional, so we therefore allow them
1769                    // to be optional here in our implementation
1770                    preceded(opt(tag(";")), opt(cfws)),
1771                    terminated(parameter, opt(cfws)),
1772                )),
1773            ),
1774        ),
1775    )
1776    .parse(input)?;
1777
1778    let mut value: BString = (*mime_type).into();
1779    value.push_char('/');
1780    value.push_str(mime_subtype);
1781
1782    Ok((loc, MimeParameters { value, parameters }))
1783}
1784
1785fn content_transfer_encoding(input: Span) -> IResult<Span, MimeParameters> {
1786    let (loc, (value, _, parameters)) = context(
1787        "content_transfer_encoding",
1788        preceded(
1789            opt(cfws),
1790            (
1791                mime_token,
1792                opt(cfws),
1793                many0(preceded(
1794                    // Note that RFC 2231 is a bit of a mess, showing examples
1795                    // without `;` as a separator in the original text, but
1796                    // in the errata from several years later, corrects those
1797                    // to show the `;`.
1798                    // In the meantime, there are implementations that assume
1799                    // that the `;` is optional, so we therefore allow them
1800                    // to be optional here in our implementation
1801                    preceded(opt(tag(";")), opt(cfws)),
1802                    terminated(parameter, opt(cfws)),
1803                )),
1804            ),
1805        ),
1806    )
1807    .parse(input)?;
1808
1809    Ok((
1810        loc,
1811        MimeParameters {
1812            value: value.as_bytes().into(),
1813            parameters,
1814        },
1815    ))
1816}
1817
1818// parameter = { regular_parameter | extended_parameter }
1819fn parameter(input: Span) -> IResult<Span, MimeParameter> {
1820    context(
1821        "parameter",
1822        alt((
1823            // Note that RFC2047 explicitly prohibits both of
1824            // these 2047 cases from appearing here, but that
1825            // major MUAs produce this sort of prohibited content
1826            // and we thus need to accommodate it
1827            param_with_unquoted_rfc2047,
1828            param_with_quoted_rfc2047,
1829            regular_parameter,
1830            extended_param_with_charset,
1831            extended_param_no_charset,
1832        )),
1833    )
1834    .parse(input)
1835}
1836
1837fn param_with_unquoted_rfc2047(input: Span) -> IResult<Span, MimeParameter> {
1838    context(
1839        "param_with_unquoted_rfc2047",
1840        map(
1841            (attribute, opt(cfws), tag("="), opt(cfws), encoded_word),
1842            |(name, _, _, _, value)| MimeParameter {
1843                name: name.as_bytes().into(),
1844                value: value.as_bytes().into(),
1845                section: None,
1846                encoding: MimeParameterEncoding::UnquotedRfc2047,
1847                mime_charset: None,
1848                mime_language: None,
1849            },
1850        ),
1851    )
1852    .parse(input)
1853}
1854
1855fn param_with_quoted_rfc2047(input: Span) -> IResult<Span, MimeParameter> {
1856    context(
1857        "param_with_quoted_rfc2047",
1858        map(
1859            (
1860                attribute,
1861                opt(cfws),
1862                tag("="),
1863                opt(cfws),
1864                delimited(tag("\""), encoded_word, tag("\"")),
1865            ),
1866            |(name, _, _, _, value)| MimeParameter {
1867                name: name.as_bytes().into(),
1868                value: value.as_bytes().into(),
1869                section: None,
1870                encoding: MimeParameterEncoding::QuotedRfc2047,
1871                mime_charset: None,
1872                mime_language: None,
1873            },
1874        ),
1875    )
1876    .parse(input)
1877}
1878
1879fn extended_param_with_charset(input: Span) -> IResult<Span, MimeParameter> {
1880    context(
1881        "extended_param_with_charset",
1882        map(
1883            (
1884                attribute,
1885                opt(section),
1886                tag("*"),
1887                opt(cfws),
1888                tag("="),
1889                opt(cfws),
1890                opt(mime_charset),
1891                tag("'"),
1892                opt(mime_language),
1893                tag("'"),
1894                map(
1895                    recognize(many0(alt((ext_octet, take_while1(is_attribute_char))))),
1896                    |s: Span| (*s).into(),
1897                ),
1898            ),
1899            |(name, section, _, _, _, _, mime_charset, _, mime_language, _, value)| MimeParameter {
1900                name: name.as_bytes().into(),
1901                section,
1902                mime_charset: mime_charset.map(|s| s.as_bytes().into()),
1903                mime_language: mime_language.map(|s| s.as_bytes().into()),
1904                encoding: MimeParameterEncoding::Rfc2231,
1905                value,
1906            },
1907        ),
1908    )
1909    .parse(input)
1910}
1911
1912fn extended_param_no_charset(input: Span) -> IResult<Span, MimeParameter> {
1913    context(
1914        "extended_param_no_charset",
1915        map(
1916            (
1917                attribute,
1918                opt(section),
1919                opt(tag("*")),
1920                opt(cfws),
1921                tag("="),
1922                opt(cfws),
1923                alt((
1924                    quoted_string,
1925                    map(
1926                        recognize(many0(alt((ext_octet, take_while1(is_attribute_char))))),
1927                        |s: Span| (*s).into(),
1928                    ),
1929                )),
1930            ),
1931            |(name, section, star, _, _, _, value)| MimeParameter {
1932                name: name.as_bytes().into(),
1933                section,
1934                mime_charset: None,
1935                mime_language: None,
1936                encoding: if star.is_some() {
1937                    MimeParameterEncoding::Rfc2231
1938                } else {
1939                    MimeParameterEncoding::None
1940                },
1941                value,
1942            },
1943        ),
1944    )
1945    .parse(input)
1946}
1947
1948fn mime_charset(input: Span) -> IResult<Span, Span> {
1949    context(
1950        "mime_charset",
1951        take_while1(|c| is_mime_token(c) && c != b'\''),
1952    )
1953    .parse(input)
1954}
1955
1956fn mime_language(input: Span) -> IResult<Span, Span> {
1957    context(
1958        "mime_language",
1959        take_while1(|c| is_mime_token(c) && c != b'\''),
1960    )
1961    .parse(input)
1962}
1963
1964fn ext_octet(input: Span) -> IResult<Span, Span> {
1965    context(
1966        "ext_octet",
1967        recognize((
1968            tag("%"),
1969            take_while_m_n(2, 2, |b: u8| b.is_ascii_hexdigit()),
1970        )),
1971    )
1972    .parse(input)
1973}
1974
1975// section = { "*" ~ ASCII_DIGIT+ }
1976fn section(input: Span) -> IResult<Span, u32> {
1977    context("section", preceded(tag("*"), nom::character::complete::u32)).parse(input)
1978}
1979
1980// regular_parameter = { attribute ~ cfws? ~ "=" ~ cfws? ~ value }
1981fn regular_parameter(input: Span) -> IResult<Span, MimeParameter> {
1982    context(
1983        "regular_parameter",
1984        map(
1985            (attribute, opt(cfws), tag("="), opt(cfws), value),
1986            |(name, _, _, _, value)| MimeParameter {
1987                name: name.as_bytes().into(),
1988                value: value.as_bytes().into(),
1989                section: None,
1990                encoding: MimeParameterEncoding::None,
1991                mime_charset: None,
1992                mime_language: None,
1993            },
1994        ),
1995    )
1996    .parse(input)
1997}
1998
1999// attribute = { attribute_char+ }
2000// attribute_char = { !(" " | ctl | tspecials | "*" | "'" | "%") ~ char }
2001fn attribute(input: Span) -> IResult<Span, Span> {
2002    context("attribute", take_while1(is_attribute_char)).parse(input)
2003}
2004
2005fn value(input: Span) -> IResult<Span, BString> {
2006    context(
2007        "value",
2008        alt((map(mime_token, |s: Span| (*s).into()), quoted_string)),
2009    )
2010    .parse(input)
2011}
2012
2013pub struct Parser;
2014
2015impl Parser {
2016    pub fn parse_mailbox_list_header(text: &[u8]) -> Result<MailboxList> {
2017        parse_with(text, mailbox_list)
2018    }
2019
2020    pub fn parse_mailbox_header(text: &[u8]) -> Result<Mailbox> {
2021        parse_with(text, mailbox)
2022    }
2023
2024    pub fn parse_address_list_header(text: &[u8]) -> Result<AddressList> {
2025        parse_with(text, address_list)
2026    }
2027
2028    pub fn parse_msg_id_header(text: &[u8]) -> Result<MessageID> {
2029        parse_with(text, msg_id)
2030    }
2031
2032    pub fn parse_msg_id_header_list(text: &[u8]) -> Result<Vec<MessageID>> {
2033        parse_with(text, msg_id_list)
2034    }
2035
2036    pub fn parse_content_id_header(text: &[u8]) -> Result<MessageID> {
2037        parse_with(text, content_id)
2038    }
2039
2040    pub fn parse_content_type_header(text: &[u8]) -> Result<MimeParameters> {
2041        parse_with(text, content_type)
2042    }
2043
2044    pub fn parse_content_transfer_encoding_header(text: &[u8]) -> Result<MimeParameters> {
2045        parse_with(text, content_transfer_encoding)
2046    }
2047
2048    pub fn parse_unstructured_header(text: &[u8]) -> Result<BString> {
2049        parse_with(text, unstructured)
2050    }
2051
2052    pub fn parse_authentication_results_header(text: &[u8]) -> Result<AuthenticationResults> {
2053        parse_with(text, authentication_results)
2054    }
2055
2056    pub fn parse_arc_authentication_results_header(
2057        text: &[u8],
2058    ) -> Result<ARCAuthenticationResults> {
2059        parse_with(text, arc_authentication_results)
2060    }
2061}
2062
2063#[serde_as]
2064#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2065#[serde(deny_unknown_fields)]
2066pub struct ARCAuthenticationResults {
2067    pub instance: u8,
2068    #[serde_as(as = "BStringUtf8")]
2069    pub serv_id: BString,
2070    pub version: Option<u32>,
2071    pub results: Vec<AuthenticationResult>,
2072}
2073
2074impl EncodeHeaderValue for ARCAuthenticationResults {
2075    fn encode_value(&self) -> SharedString<'static> {
2076        let mut result = format!("i={}; ", self.instance).into_bytes();
2077
2078        emit_value_token(&self.serv_id, &mut result);
2079        if let Some(v) = self.version {
2080            result.push_str(format!(" {v}"));
2081        }
2082
2083        if self.results.is_empty() {
2084            result.push_str("; none");
2085        } else {
2086            for res in &self.results {
2087                result.push_str(";\r\n\t");
2088                emit_value_token(res.method.as_bytes(), &mut result);
2089                if let Some(v) = res.method_version {
2090                    result.push_str(format!("/{v}"));
2091                }
2092                result.push(b'=');
2093                emit_value_token(res.result.as_bytes(), &mut result);
2094                if let Some(reason) = &res.reason {
2095                    result.push_str(" reason=");
2096                    emit_value_token(reason.as_bytes(), &mut result);
2097                }
2098                for (k, v) in &res.props {
2099                    // Skip a key that sanitizes to nothing. Emitting `=value`
2100                    // with no key would be a malformed (though not injectable)
2101                    // value.
2102                    if !k.chars().any(is_prop_key_char) {
2103                        continue;
2104                    }
2105                    result.push_str("\r\n\t");
2106                    emit_prop_key(k, &mut result);
2107                    result.push(b'=');
2108                    emit_value_token(v.as_bytes(), &mut result);
2109                }
2110            }
2111        }
2112
2113        result.into()
2114    }
2115}
2116
2117#[serde_as]
2118#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2119#[serde(deny_unknown_fields)]
2120pub struct AuthenticationResults {
2121    #[serde_as(as = "BStringUtf8")]
2122    pub serv_id: BString,
2123    #[serde(default)]
2124    pub version: Option<u32>,
2125    #[serde(default)]
2126    pub results: Vec<AuthenticationResult>,
2127}
2128
2129/// Emits an Authentication-Results value into target, quoting it when it
2130/// contains anything outside the mime-token set, and dropping control
2131/// characters.
2132fn emit_value_token(value: &[u8], target: &mut Vec<u8>) {
2133    // Allow '@' bare since the pvalue parser handles @domain and local@domain
2134    let use_quoted_string = !value.iter().all(|&c| is_mime_token(c) || c == b'@');
2135    if use_quoted_string {
2136        target.push(b'"');
2137        for (start, end, c) in value.char_indices() {
2138            // Drop control characters other than HTAB: a bare CR or LF inside a
2139            // quoted-string ends the header line, and a sender-controlled value
2140            // could use that to inject further lines beneath ours. HTAB is
2141            // legal FWS inside a quoted-string, so it is preserved. A raw
2142            // control byte decodes via char_indices to its own ASCII character,
2143            // so it is caught here rather than as invalid UTF-8.
2144            if c.is_control() && c != '\t' {
2145                continue;
2146            }
2147            if c == '"' || c == '\\' {
2148                target.push(b'\\');
2149            }
2150            target.push_str(&value[start..end]);
2151        }
2152        target.push(b'"');
2153    } else {
2154        target.push_str(value);
2155    }
2156}
2157
2158/// Returns true when the character is one an RFC 8601 property key may contain:
2159/// ASCII alphanumerics, `-`, and `.`.
2160fn is_prop_key_char(c: char) -> bool {
2161    c.is_ascii_alphanumeric() || c == '-' || c == '.'
2162}
2163
2164/// Emits a property key (`ptype.property`) into target, keeping only the
2165/// characters a key may contain. A key is always emitted unquoted. Any other
2166/// byte is dropped, including a control character from a Lua-supplied key.
2167fn emit_prop_key(key: &str, target: &mut Vec<u8>) {
2168    for c in key.chars() {
2169        if is_prop_key_char(c) {
2170            target.push(c as u8);
2171        }
2172    }
2173}
2174
2175impl EncodeHeaderValue for AuthenticationResults {
2176    fn encode_value(&self) -> SharedString<'static> {
2177        let mut result = Vec::new();
2178        emit_value_token(&self.serv_id, &mut result);
2179        if let Some(v) = self.version {
2180            result.push_str(format!(" {v}"));
2181        }
2182        if self.results.is_empty() {
2183            result.push_str("; none");
2184        } else {
2185            for res in &self.results {
2186                result.push_str(";\r\n\t");
2187                emit_value_token(res.method.as_bytes(), &mut result);
2188                if let Some(v) = res.method_version {
2189                    result.push_str(format!("/{v}"));
2190                }
2191                result.push(b'=');
2192                emit_value_token(res.result.as_bytes(), &mut result);
2193                if let Some(reason) = &res.reason {
2194                    result.push_str(" reason=");
2195                    emit_value_token(reason.as_bytes(), &mut result);
2196                }
2197                for (k, v) in &res.props {
2198                    // Skip a key that sanitizes to nothing. Emitting `=value`
2199                    // with no key would be a malformed (though not injectable)
2200                    // value.
2201                    if !k.chars().any(is_prop_key_char) {
2202                        continue;
2203                    }
2204                    result.push_str("\r\n\t");
2205                    emit_prop_key(k, &mut result);
2206                    result.push(b'=');
2207                    emit_value_token(v.as_bytes(), &mut result);
2208                }
2209            }
2210        }
2211
2212        result.into()
2213    }
2214}
2215
2216#[serde_as]
2217#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2218#[serde(deny_unknown_fields)]
2219pub struct AuthenticationResult {
2220    pub method: String,
2221    #[serde(default)]
2222    pub method_version: Option<u32>,
2223    pub result: String,
2224    #[serde_as(as = "Option<BStringUtf8>")]
2225    #[serde(default)]
2226    pub reason: Option<BString>,
2227    #[serde_as(as = "BTreeMap<_, BStringUtf8>")]
2228    #[serde(default)]
2229    pub props: BTreeMap<String, BString>,
2230}
2231
2232#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2233#[serde(deny_unknown_fields)]
2234pub struct AddrSpec {
2235    pub local_part: String,
2236    pub domain: String,
2237}
2238
2239impl AddrSpec {
2240    pub fn new(local_part: &str, domain: &str) -> Self {
2241        Self {
2242            local_part: local_part.into(),
2243            domain: domain.into(),
2244        }
2245    }
2246
2247    pub fn parse(email: &str) -> Result<Self> {
2248        parse_with(email.as_bytes(), addr_spec)
2249    }
2250}
2251
2252impl EncodeHeaderValue for AddrSpec {
2253    fn encode_value(&self) -> SharedString<'static> {
2254        let mut result: Vec<u8> = vec![];
2255
2256        let needs_quoting = !self
2257            .local_part
2258            .as_bytes()
2259            .iter()
2260            .all(|&c| is_atext(c) || c == b'.');
2261        if needs_quoting {
2262            result.push(b'"');
2263            // RFC5321 4.1.2 qtextSMTP:
2264            // within a quoted string, any ASCII graphic or space is permitted without
2265            // blackslash-quoting except double-quote and the backslash itself.
2266
2267            for &c in self.local_part.as_bytes().iter() {
2268                if c == b'"' || c == b'\\' {
2269                    result.push(b'\\');
2270                }
2271                result.push(c);
2272            }
2273            result.push(b'"');
2274        } else {
2275            result.push_str(&self.local_part);
2276        }
2277        result.push(b'@');
2278        result.push_str(&self.domain);
2279
2280        result.into()
2281    }
2282}
2283
2284#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2285#[serde(untagged)]
2286pub enum Address {
2287    Mailbox(Mailbox),
2288    Group { name: String, entries: MailboxList },
2289}
2290
2291#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2292#[serde(deny_unknown_fields, transparent)]
2293pub struct AddressList(pub Vec<Address>);
2294
2295impl std::ops::Deref for AddressList {
2296    type Target = Vec<Address>;
2297    fn deref(&self) -> &Vec<Address> {
2298        &self.0
2299    }
2300}
2301
2302impl AddressList {
2303    pub fn extract_first_mailbox(&self) -> Option<&Mailbox> {
2304        let address = self.0.first()?;
2305        match address {
2306            Address::Mailbox(mailbox) => Some(mailbox),
2307            Address::Group { entries, .. } => entries.extract_first_mailbox(),
2308        }
2309    }
2310}
2311
2312#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2313#[serde(deny_unknown_fields, transparent)]
2314pub struct MailboxList(pub Vec<Mailbox>);
2315
2316impl std::ops::Deref for MailboxList {
2317    type Target = Vec<Mailbox>;
2318    fn deref(&self) -> &Vec<Mailbox> {
2319        &self.0
2320    }
2321}
2322
2323impl MailboxList {
2324    pub fn extract_first_mailbox(&self) -> Option<&Mailbox> {
2325        self.0.first()
2326    }
2327}
2328
2329#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2330#[serde(deny_unknown_fields)]
2331pub struct Mailbox {
2332    pub name: Option<String>,
2333    pub address: AddrSpec,
2334}
2335
2336#[serde_as]
2337#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
2338#[serde(transparent)]
2339pub struct MessageID(#[serde_as(as = "BStringUtf8")] pub BString);
2340
2341impl EncodeHeaderValue for MessageID {
2342    fn encode_value(&self) -> SharedString<'static> {
2343        let mut result = Vec::<u8>::with_capacity(self.0.len() + 2);
2344        result.push(b'<');
2345        result.push_str(&self.0);
2346        result.push(b'>');
2347        result.into()
2348    }
2349}
2350
2351impl EncodeHeaderValue for Vec<MessageID> {
2352    fn encode_value(&self) -> SharedString<'static> {
2353        let mut result = BString::default();
2354        for id in self {
2355            if !result.is_empty() {
2356                result.push_str("\r\n\t");
2357            }
2358            result.push(b'<');
2359            result.push_str(&id.0);
2360            result.push(b'>');
2361        }
2362        result.into()
2363    }
2364}
2365
2366// In theory, everyone would be aware of RFC 2231 and we can stop here,
2367// but in practice, things are messy.  At some point someone started
2368// to emit encoded-words insides quoted-string values, and for the sake
2369// of compatibility what we see now is technically illegal stuff like
2370// Content-Disposition: attachment; filename="=?UTF-8?B?5pel5pys6Kqe44Gu5re75LuY?="
2371// being used to represent UTF-8 filenames.
2372// As such, in our RFC 2231 handling, we also need to accommodate
2373// these bogus representations, hence their presence in this enum
2374#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2375pub(crate) enum MimeParameterEncoding {
2376    None,
2377    Rfc2231,
2378    UnquotedRfc2047,
2379    QuotedRfc2047,
2380}
2381
2382#[derive(Debug, Clone, PartialEq, Eq)]
2383struct MimeParameter {
2384    pub name: BString,
2385    pub section: Option<u32>,
2386    pub mime_charset: Option<BString>,
2387    pub mime_language: Option<BString>,
2388    pub encoding: MimeParameterEncoding,
2389    pub value: BString,
2390}
2391
2392#[derive(Debug, Clone, PartialEq, Eq)]
2393pub struct MimeParameters {
2394    pub value: BString,
2395    parameters: Vec<MimeParameter>,
2396}
2397
2398impl MimeParameters {
2399    pub fn new(value: impl AsRef<[u8]>) -> Self {
2400        Self {
2401            value: value.as_ref().into(),
2402            parameters: vec![],
2403        }
2404    }
2405
2406    /// Decode all named parameters per RFC 2231 and return a map
2407    /// of the parameter names to parameters values.
2408    /// Incorrectly encoded parameters are silently ignored
2409    /// and are not returned in the resulting map.
2410    pub fn parameter_map(&self) -> BTreeMap<BString, BString> {
2411        self.grouped_parameters().into_values().collect()
2412    }
2413
2414    /// Returns each distinct parameter name mapped to the original spelling
2415    /// of its first occurrence and its decoded value. Keyed by the lowercased
2416    /// name to fold case-insensitive duplicates together. Grouping in one pass
2417    /// keeps this linearithmic rather than quadratic in the parameter count.
2418    fn grouped_parameters(&self) -> BTreeMap<BString, (BString, BString)> {
2419        let mut groups: BTreeMap<BString, (BString, Vec<&MimeParameter>)> = BTreeMap::new();
2420        for entry in &self.parameters {
2421            let folded: BString = entry.name.to_ascii_lowercase().into();
2422            groups
2423                .entry(folded)
2424                .or_insert_with(|| (entry.name.clone(), vec![]))
2425                .1
2426                .push(entry);
2427        }
2428        groups
2429            .into_iter()
2430            .map(|(folded, (display_name, elements))| {
2431                (
2432                    folded,
2433                    (display_name, Self::decode_parameter_elements(elements)),
2434                )
2435            })
2436            .collect()
2437    }
2438
2439    /// Insert each incoming parameter whose name is not already present
2440    /// (case-insensitively), leaving existing parameters untouched. Incoming
2441    /// values are stored verbatim with no encoding. The set of present names is
2442    /// computed once, keeping the merge linearithmic rather than quadratic in
2443    /// the combined parameter count.
2444    pub fn merge_missing_parameters(&mut self, incoming: BTreeMap<BString, BString>) {
2445        let mut present: BTreeSet<BString> = self
2446            .parameters
2447            .iter()
2448            .map(|p| p.name.to_ascii_lowercase().into())
2449            .collect();
2450        for (name, value) in incoming {
2451            let folded: BString = name.to_ascii_lowercase().into();
2452            if present.insert(folded) {
2453                self.parameters.push(MimeParameter {
2454                    name,
2455                    value,
2456                    section: None,
2457                    mime_charset: None,
2458                    mime_language: None,
2459                    encoding: MimeParameterEncoding::None,
2460                });
2461            }
2462        }
2463    }
2464
2465    /// Retrieve the value for a named parameter.
2466    /// This method will attempt to decode any %-encoded values
2467    /// per RFC 2231 and combine multi-element fields into a single
2468    /// contiguous value.
2469    /// Invalid charsets and encoding will be silently ignored.
2470    pub fn get(&self, name: impl AsRef<[u8]>) -> Option<BString> {
2471        let name = name.as_ref();
2472        let elements: Vec<_> = self
2473            .parameters
2474            .iter()
2475            .filter(|p| p.name.eq_ignore_ascii_case(name))
2476            .collect();
2477        if elements.is_empty() {
2478            return None;
2479        }
2480        Some(Self::decode_parameter_elements(elements))
2481    }
2482
2483    /// Decode a group of parameter elements that share a name into a value,
2484    /// ordering multi-part (RFC 2231 sectioned) elements by section and
2485    /// applying any %-encoding. Invalid charsets and encodings are silently
2486    /// ignored.
2487    ///
2488    /// A well-formed parameter names each RFC 2231 continuation section at most
2489    /// once (RFC 2231 s3), and a simple parameter (no section) appears once
2490    /// (RFC 2045 s5.1). A repeated section, or a repeated simple parameter, is
2491    /// malformed. The last occurrence wins. `elements` is taken in document
2492    /// order so that last is the one appearing latest in the header.
2493    fn decode_parameter_elements(elements: Vec<&MimeParameter>) -> BString {
2494        // Deduplicate by section, keeping the last occurrence, and order by
2495        // section. A BTreeMap keyed on Option<u32> orders None (the simple,
2496        // unsectioned form) before the numerically-ordered sections.
2497        let elements: Vec<&MimeParameter> = elements
2498            .into_iter()
2499            .map(|ele| (ele.section, ele))
2500            .collect::<BTreeMap<_, _>>()
2501            .into_values()
2502            .collect();
2503
2504        let mut mime_charset = None;
2505        let mut result: Vec<u8> = vec![];
2506
2507        for ele in elements {
2508            if let Some(cset) = ele.mime_charset.as_ref().and_then(|b| b.to_str().ok()) {
2509                mime_charset = Encoding::by_name(&*cset);
2510            }
2511
2512            match ele.encoding {
2513                MimeParameterEncoding::Rfc2231 => {
2514                    if let Some(charset) = mime_charset.as_ref() {
2515                        let mut chars = ele.value.chars();
2516                        let mut bytes: Vec<u8> = vec![];
2517
2518                        fn char_to_bytes(c: char, bytes: &mut Vec<u8>) {
2519                            let mut buf = [0u8; 8];
2520                            let s = c.encode_utf8(&mut buf);
2521                            for b in s.bytes() {
2522                                bytes.push(b);
2523                            }
2524                        }
2525
2526                        'next_char: while let Some(c) = chars.next() {
2527                            match c {
2528                                '%' => {
2529                                    let mut value = 0u8;
2530                                    for _ in 0..2 {
2531                                        match chars.next() {
2532                                            Some(n) => match n {
2533                                                '0'..='9' => {
2534                                                    value <<= 4;
2535                                                    value |= n as u32 as u8 - b'0';
2536                                                }
2537                                                'a'..='f' => {
2538                                                    value <<= 4;
2539                                                    value |= (n as u32 as u8 - b'a') + 10;
2540                                                }
2541                                                'A'..='F' => {
2542                                                    value <<= 4;
2543                                                    value |= (n as u32 as u8 - b'A') + 10;
2544                                                }
2545                                                _ => {
2546                                                    char_to_bytes('%', &mut bytes);
2547                                                    char_to_bytes(n, &mut bytes);
2548                                                    break 'next_char;
2549                                                }
2550                                            },
2551                                            None => {
2552                                                char_to_bytes('%', &mut bytes);
2553                                                break 'next_char;
2554                                            }
2555                                        }
2556                                    }
2557
2558                                    bytes.push(value);
2559                                }
2560                                c => {
2561                                    char_to_bytes(c, &mut bytes);
2562                                }
2563                            }
2564                        }
2565
2566                        if let Ok(decoded) = charset.decode_simple(&bytes) {
2567                            result.push_str(&decoded);
2568                        }
2569                    } else {
2570                        result.push_str(&ele.value);
2571                    }
2572                }
2573                MimeParameterEncoding::UnquotedRfc2047
2574                | MimeParameterEncoding::QuotedRfc2047
2575                | MimeParameterEncoding::None => {
2576                    result.push_str(&ele.value);
2577                }
2578            }
2579        }
2580
2581        result.into()
2582    }
2583
2584    /// Remove the named parameter
2585    pub fn remove(&mut self, name: impl AsRef<[u8]>) {
2586        let name = name.as_ref();
2587        self.parameters
2588            .retain(|p| !p.name.eq_ignore_ascii_case(name));
2589    }
2590
2591    pub fn set(&mut self, name: impl AsRef<[u8]>, value: impl AsRef<[u8]>) {
2592        self.set_with_encoding(name, value, MimeParameterEncoding::None)
2593    }
2594
2595    pub(crate) fn set_with_encoding(
2596        &mut self,
2597        name: impl AsRef<[u8]>,
2598        value: impl AsRef<[u8]>,
2599        encoding: MimeParameterEncoding,
2600    ) {
2601        self.remove(name.as_ref());
2602
2603        self.parameters.push(MimeParameter {
2604            name: name.as_ref().into(),
2605            value: value.as_ref().into(),
2606            section: None,
2607            mime_charset: None,
2608            mime_language: None,
2609            encoding,
2610        });
2611    }
2612
2613    pub fn is_multipart(&self) -> bool {
2614        self.value.starts_with_str("message/") || self.value.starts_with_str("multipart/")
2615    }
2616
2617    pub fn is_text(&self) -> bool {
2618        self.value.starts_with_str("text/")
2619    }
2620}
2621
2622impl EncodeHeaderValue for MimeParameters {
2623    fn encode_value(&self) -> SharedString<'static> {
2624        let mut result = self.value.clone();
2625        let grouped = self.grouped_parameters();
2626        let names: BTreeMap<&BStr, MimeParameterEncoding> = self
2627            .parameters
2628            .iter()
2629            .map(|p| (p.name.as_bstr(), p.encoding))
2630            .collect();
2631
2632        for (name, stated_encoding) in names {
2633            let folded: BString = name.to_ascii_lowercase().into();
2634            let value = grouped
2635                .get(&folded)
2636                .map(|(_display_name, value)| value.clone())
2637                .expect("name to be present");
2638
2639            match stated_encoding {
2640                MimeParameterEncoding::UnquotedRfc2047 => {
2641                    let encoded = qp_encode(&value);
2642                    result.push_str(format!(";\r\n\t{name}={encoded}"));
2643                }
2644                MimeParameterEncoding::QuotedRfc2047 => {
2645                    let encoded = qp_encode(&value);
2646                    result.push_str(format!(";\r\n\t{name}=\"{encoded}\""));
2647                }
2648                MimeParameterEncoding::None | MimeParameterEncoding::Rfc2231 => {
2649                    let needs_encoding = value.iter().any(|&c| !is_mime_token(c) || !c.is_ascii());
2650                    // Prefer to use quoted_string representation when possible, as it doesn't
2651                    // require any RFC 2231 encoding
2652                    let use_quoted_string = value
2653                        .iter()
2654                        .all(|&c| (is_qtext(c) || is_quoted_pair(c)) && c.is_ascii());
2655
2656                    let mut params = vec![];
2657                    let mut chars = value.char_indices().peekable();
2658                    while chars.peek().is_some() {
2659                        let count = params.len();
2660                        let is_first = count == 0;
2661                        let prefix = if use_quoted_string {
2662                            "\""
2663                        } else if is_first && needs_encoding {
2664                            "UTF-8''"
2665                        } else {
2666                            ""
2667                        };
2668                        // A parameter name longer than the fold target makes
2669                        // the framing wider than the target. Saturate to zero
2670                        // rather than underflow. The loop below always consumes
2671                        // at least one character per line, keeping progress
2672                        // even when the budget is zero.
2673                        let limit = 74usize.saturating_sub(name.len() + 4 + prefix.len());
2674
2675                        let mut encoded: Vec<u8> = vec![];
2676
2677                        loop {
2678                            let Some((start, end, c)) = chars.next() else {
2679                                break;
2680                            };
2681                            let s = &value[start..end];
2682
2683                            if use_quoted_string {
2684                                if c == '"' || c == '\\' {
2685                                    encoded.push(b'\\');
2686                                }
2687                                encoded.push_str(s);
2688                            } else if (c as u32) <= 0xff
2689                                && is_mime_token(c as u32 as u8)
2690                                && (!needs_encoding || c != '%')
2691                            {
2692                                encoded.push_str(s);
2693                            } else {
2694                                for b in s.bytes() {
2695                                    encoded.push(b'%');
2696                                    encoded.push(HEX_CHARS[(b as usize) >> 4]);
2697                                    encoded.push(HEX_CHARS[(b as usize) & 0x0f]);
2698                                }
2699                            }
2700
2701                            if encoded.len() >= limit {
2702                                break;
2703                            }
2704                        }
2705
2706                        if use_quoted_string {
2707                            encoded.push(b'"');
2708                        }
2709
2710                        params.push(MimeParameter {
2711                            name: name.into(),
2712                            section: Some(count as u32),
2713                            mime_charset: if is_first { Some("UTF-8".into()) } else { None },
2714                            mime_language: None,
2715                            encoding: if needs_encoding {
2716                                MimeParameterEncoding::Rfc2231
2717                            } else {
2718                                MimeParameterEncoding::None
2719                            },
2720                            value: encoded.into(),
2721                        })
2722                    }
2723                    if params.len() == 1 {
2724                        params.last_mut().map(|p| p.section = None);
2725                    }
2726                    for p in params {
2727                        result.push_str(";\r\n\t");
2728                        let charset_tick = if !use_quoted_string
2729                            && (p.mime_charset.is_some() || p.mime_language.is_some())
2730                        {
2731                            "'"
2732                        } else {
2733                            ""
2734                        };
2735                        let lang_tick = if !use_quoted_string
2736                            && (p.mime_language.is_some() || p.mime_charset.is_some())
2737                        {
2738                            "'"
2739                        } else {
2740                            ""
2741                        };
2742
2743                        let section = p
2744                            .section
2745                            .map(|s| format!("*{s}"))
2746                            .unwrap_or_else(String::new);
2747
2748                        let uses_encoding =
2749                            if !use_quoted_string && p.encoding == MimeParameterEncoding::Rfc2231 {
2750                                "*"
2751                            } else {
2752                                ""
2753                            };
2754                        let charset = if use_quoted_string {
2755                            BStr::new("\"")
2756                        } else {
2757                            p.mime_charset
2758                                .as_ref()
2759                                .map(|b| b.as_bstr())
2760                                .unwrap_or(BStr::new(""))
2761                        };
2762                        let lang = p
2763                            .mime_language
2764                            .as_ref()
2765                            .map(|b| b.as_bstr())
2766                            .unwrap_or(BStr::new(""));
2767
2768                        let line = format!(
2769                            "{name}{section}{uses_encoding}={charset}{charset_tick}{lang}{lang_tick}{value}",
2770                            name = &p.name,
2771                            value = &p.value
2772                        );
2773                        result.push_str(&line);
2774                    }
2775                }
2776            }
2777        }
2778        result.into()
2779    }
2780}
2781
2782static HEX_CHARS: &[u8] = b"0123456789ABCDEF";
2783
2784pub(crate) fn qp_encode(s: &[u8]) -> String {
2785    let prefix = b"=?UTF-8?q?";
2786    let suffix = b"?=";
2787    let limit = 72 - (prefix.len() + suffix.len());
2788
2789    let mut result = Vec::with_capacity(s.len());
2790
2791    result.extend_from_slice(prefix);
2792    let mut line_length = 0;
2793
2794    enum Bytes<'a> {
2795        Passthru(&'a [u8]),
2796        Encode(&'a [u8]),
2797    }
2798
2799    // Iterate by char so that we don't confuse space (0x20) with a
2800    // utf8 subsequence and incorrectly encode the input string.
2801    for (start, end, c) in s.char_indices() {
2802        let bytes = &s[start..end];
2803
2804        // RFC 2047 section 5(3) restricts the punctuation allowed unencoded in
2805        // a Q encoded-word within a phrase to this set. Because a phrase is the
2806        // most restrictive context this encoder serves, encoding to it keeps
2807        // one encoder valid everywhere, at the cost of encoding some
2808        // punctuation a Subject could have left alone. Since an underscore
2809        // represents a space in the Q encoding, a literal underscore must be
2810        // encoded as =5F to avoid a decoder turning it back into a space.
2811        let b = if c.is_ascii_alphanumeric() || matches!(c, '!' | '*' | '+' | '-' | '/') {
2812            Bytes::Passthru(bytes)
2813        } else if c == ' ' {
2814            Bytes::Passthru(b"_")
2815        } else {
2816            Bytes::Encode(bytes)
2817        };
2818
2819        let need_len = match b {
2820            Bytes::Passthru(b) => b.len(),
2821            Bytes::Encode(b) => b.len() * 3,
2822        };
2823
2824        if need_len > limit - line_length {
2825            // Need to wrap
2826            result.extend_from_slice(suffix);
2827            result.extend_from_slice(b"\r\n\t");
2828            result.extend_from_slice(prefix);
2829            line_length = 0;
2830        }
2831
2832        match b {
2833            Bytes::Passthru(c) => {
2834                result.extend_from_slice(c);
2835            }
2836            Bytes::Encode(bytes) => {
2837                for &c in bytes {
2838                    result.push(b'=');
2839                    result.push(HEX_CHARS[(c as usize) >> 4]);
2840                    result.push(HEX_CHARS[(c as usize) & 0x0f]);
2841                }
2842            }
2843        }
2844
2845        line_length += need_len;
2846    }
2847
2848    if line_length > 0 {
2849        result.extend_from_slice(suffix);
2850    }
2851
2852    // Safety: we ensured that everything we output is in the ASCII
2853    // range, therefore the string is valid UTF-8
2854    unsafe { String::from_utf8_unchecked(result) }
2855}
2856
2857#[cfg(test)]
2858#[test]
2859fn test_qp_encode() {
2860    let encoded = qp_encode(
2861        b"hello, I am a line that is this long, or maybe a little \
2862        bit longer than this, and that should get wrapped by the encoder",
2863    );
2864    k9::snapshot!(
2865        encoded,
2866        r#"
2867=?UTF-8?q?hello=2C_I_am_a_line_that_is_this_long=2C_or_maybe_a_little_?=\r
2868\t=?UTF-8?q?bit_longer_than_this=2C_and_that_should_get_wrapped_by_the_e?=\r
2869\t=?UTF-8?q?ncoder?=
2870"#
2871    );
2872}
2873
2874#[cfg(test)]
2875#[test]
2876fn test_qp_encode_literal_underscore() {
2877    // A literal underscore must be escaped as =5F to distinguish it from the
2878    // underscore that Q encoding uses to represent a space.
2879    let encoded = qp_encode("formul\u{e1}rios Word_TEST".as_bytes());
2880    k9::assert_equal!(encoded, "=?UTF-8?q?formul=C3=A1rios_Word=5FTEST?=");
2881}
2882
2883/// Quote input string `s`, using a backslash escape, if any
2884/// of the characters is NOT atext.  When quoting, the input
2885/// string is enclosed in quotes.
2886fn quote_string(s: impl AsRef<[u8]>) -> BString {
2887    let s = s.as_ref();
2888
2889    if s.iter().any(|&c| !is_atext(c)) {
2890        let mut result = Vec::<u8>::with_capacity(s.len() + 4);
2891        result.push(b'"');
2892        for (start, end, c) in s.char_indices() {
2893            let c = c as u32;
2894            if c <= 0xff {
2895                let c = c as u8;
2896                if c == b'\r' || c == b'\n' {
2897                    // A CR/LF that is part of a legal RFC 5322 fold (a CR?LF
2898                    // immediately followed by WSP) is kept: it is valid header
2899                    // structure, not injection. A bare CR/LF is rewritten to a
2900                    // space so it cannot terminate the header line and let the
2901                    // bytes after it be read as a separate, spurious header.
2902                    let is_fold = match c {
2903                        b'\r' => matches!(&s[end..], [b'\n', b' ' | b'\t', ..]),
2904                        _ => matches!(&s[end..], [b' ' | b'\t', ..]),
2905                    };
2906                    if is_fold {
2907                        result.push_str(&s[start..end]);
2908                    } else {
2909                        result.push(b' ');
2910                    }
2911                    continue;
2912                }
2913                if !c.is_ascii_whitespace() && !is_qtext(c) && !is_atext(c) {
2914                    result.push(b'\\');
2915                }
2916            }
2917            result.push_str(&s[start..end]);
2918        }
2919        result.push(b'"');
2920        result.into()
2921    } else {
2922        s.into()
2923    }
2924}
2925
2926#[cfg(test)]
2927#[test]
2928fn test_quote_string() {
2929    k9::snapshot!(
2930        quote_string("TEST [ne_pas_repondre]"),
2931        r#""TEST [ne_pas_repondre]""#
2932    );
2933    k9::snapshot!(quote_string("hello"), "hello");
2934    k9::snapshot!(quote_string("hello there"), r#""hello there""#);
2935    k9::snapshot!(quote_string("hello, there"), "\"hello, there\"");
2936    k9::snapshot!(quote_string("hello \"there\""), r#""hello \\"there\\"""#);
2937    k9::snapshot!(
2938        quote_string("hello c:\\backslash"),
2939        r#""hello c:\\\\backslash""#
2940    );
2941    k9::assert_equal!(quote_string("hello\n there"), "\"hello\n there\"");
2942}
2943
2944impl EncodeHeaderValue for Mailbox {
2945    fn encode_value(&self) -> SharedString<'static> {
2946        match &self.name {
2947            Some(name) => {
2948                // The display name (a quoted-string, or an RFC 2047
2949                // encoded-word that may itself already be multi-line) and the
2950                // `<addr>` are joined by a fold when they would overflow the
2951                // line, which is the only safe point: folding inside a quoted
2952                // display name would rewrite the display name content (a space
2953                // becomes a tab once unfolded). A name whose last line exceeds
2954                // the width is left as-is rather than corrupted by folding
2955                // inside its quoting or an encoded-word.
2956                let phrase: Vec<u8> = if name.is_ascii() {
2957                    quote_string(name).into()
2958                } else {
2959                    qp_encode(name.as_bytes()).into_bytes()
2960                };
2961
2962                let mut addr: Vec<u8> = vec![b'<'];
2963                addr.push_str(self.address.encode_value().as_bytes());
2964                addr.push(b'>');
2965
2966                // qp_encode may have already folded a long non-ASCII name into
2967                // multiple encoded-words separated by `\r\n\t`. Only the last
2968                // of those lines shares a line with `<addr>`, so measure from
2969                // the final fold when deciding whether to fold before `<addr>`.
2970                // quote_string only lets a raw `\n` through when it is part of
2971                // a legal fold (CR?LF followed by WSP), so any `\n` remaining
2972                // in `phrase` here is guaranteed to be a fold boundary, not
2973                // arbitrary content.
2974                let last_line_len = phrase
2975                    .rfind_byte(b'\n')
2976                    .map(|i| phrase.len() - (i + 1))
2977                    .unwrap_or(phrase.len());
2978
2979                let mut value = phrase;
2980                if last_line_len + 1 + addr.len() > kumo_wrap::SOFT_WIDTH {
2981                    value.push_str("\r\n\t");
2982                } else {
2983                    value.push(b' ');
2984                }
2985                value.push_str(&addr);
2986                value.into()
2987            }
2988            None => {
2989                let mut result: Vec<u8> = vec![];
2990                result.push(b'<');
2991                result.push_str(self.address.encode_value().as_bytes());
2992                result.push(b'>');
2993                result.into()
2994            }
2995        }
2996    }
2997}
2998
2999impl EncodeHeaderValue for MailboxList {
3000    fn encode_value(&self) -> SharedString<'static> {
3001        let mut result: Vec<u8> = vec![];
3002        for mailbox in &self.0 {
3003            if !result.is_empty() {
3004                result.push_str(",\r\n\t");
3005            }
3006            result.push_str(mailbox.encode_value().as_bytes());
3007        }
3008        result.into()
3009    }
3010}
3011
3012impl EncodeHeaderValue for Address {
3013    fn encode_value(&self) -> SharedString<'static> {
3014        match self {
3015            Self::Mailbox(mbox) => mbox.encode_value(),
3016            Self::Group { name, entries } => {
3017                let mut result: Vec<u8> = vec![];
3018                result.push_str(name);
3019                result.push(b':');
3020                result.push_str(entries.encode_value().as_bytes());
3021                result.push(b';');
3022                result.into()
3023            }
3024        }
3025    }
3026}
3027
3028impl EncodeHeaderValue for AddressList {
3029    fn encode_value(&self) -> SharedString<'static> {
3030        let mut result: Vec<u8> = vec![];
3031        for address in &self.0 {
3032            if !result.is_empty() {
3033                result.push_str(",\r\n\t");
3034            }
3035            result.push_str(address.encode_value().as_bytes());
3036        }
3037        result.into()
3038    }
3039}
3040
3041#[cfg(test)]
3042mod test {
3043    use super::*;
3044    use crate::{Header, MessageConformance, MimePart};
3045
3046    #[test]
3047    fn mailbox_encodes_at() {
3048        let mbox = Mailbox {
3049            name: Some("foo@bar.com".into()),
3050            address: AddrSpec {
3051                local_part: "foo".into(),
3052                domain: "bar.com".into(),
3053            },
3054        };
3055        assert_eq!(mbox.encode_value(), "\"foo@bar.com\" <foo@bar.com>");
3056    }
3057
3058    #[test]
3059    fn mailbox_list_singular() {
3060        let message = concat!(
3061            "From:  Someone (hello) <someone@example.com>, other@example.com,\n",
3062            "  \"John \\\"Smith\\\"\" (comment) \"More Quotes\" (more comment) <someone(another comment)@crazy.example.com(woot)>\n",
3063            "\n",
3064            "I am the body"
3065        );
3066        let msg = MimePart::parse(message).unwrap();
3067        let list = match msg.headers().from() {
3068            Err(err) => panic!("Doh.\n{err:#}"),
3069            Ok(list) => list,
3070        };
3071
3072        k9::snapshot!(
3073            list,
3074            r#"
3075Some(
3076    MailboxList(
3077        [
3078            Mailbox {
3079                name: Some(
3080                    "Someone",
3081                ),
3082                address: AddrSpec {
3083                    local_part: "someone",
3084                    domain: "example.com",
3085                },
3086            },
3087            Mailbox {
3088                name: None,
3089                address: AddrSpec {
3090                    local_part: "other",
3091                    domain: "example.com",
3092                },
3093            },
3094            Mailbox {
3095                name: Some(
3096                    "John "Smith" More Quotes",
3097                ),
3098                address: AddrSpec {
3099                    local_part: "someone",
3100                    domain: "crazy.example.com",
3101                },
3102            },
3103        ],
3104    ),
3105)
3106"#
3107        );
3108    }
3109
3110    #[test]
3111    fn docomo_non_compliant_localpart() {
3112        let message = "Sender: hello..there@docomo.ne.jp\n\n\n";
3113        let msg = MimePart::parse(message).unwrap();
3114        let err = msg.headers().sender().unwrap_err();
3115        k9::snapshot!(
3116            err,
3117            r#"
3118InvalidHeaderValueDuringGet {
3119    header_name: "Sender",
3120    error: HeaderParse(
3121        "Error at line 1, expected "@" but found ".":
3122hello..there@docomo.ne.jp
3123     ^___________________
3124
3125while parsing addr_spec
3126while parsing mailbox
3127",
3128    ),
3129}
3130"#
3131        );
3132    }
3133
3134    #[test]
3135    fn sender() {
3136        let message = "Sender: someone@[127.0.0.1]\n\n\n";
3137        let msg = MimePart::parse(message).unwrap();
3138        let list = match msg.headers().sender() {
3139            Err(err) => panic!("Doh.\n{err:#}"),
3140            Ok(list) => list,
3141        };
3142        k9::snapshot!(
3143            list,
3144            r#"
3145Some(
3146    Mailbox {
3147        name: None,
3148        address: AddrSpec {
3149            local_part: "someone",
3150            domain: "[127.0.0.1]",
3151        },
3152    },
3153)
3154"#
3155        );
3156    }
3157
3158    #[test]
3159    fn domain_literal() {
3160        let message = "From: someone@[127.0.0.1]\n\n\n";
3161        let msg = MimePart::parse(message).unwrap();
3162        let list = match msg.headers().from() {
3163            Err(err) => panic!("Doh.\n{err:#}"),
3164            Ok(list) => list,
3165        };
3166        k9::snapshot!(
3167            list,
3168            r#"
3169Some(
3170    MailboxList(
3171        [
3172            Mailbox {
3173                name: None,
3174                address: AddrSpec {
3175                    local_part: "someone",
3176                    domain: "[127.0.0.1]",
3177                },
3178            },
3179        ],
3180    ),
3181)
3182"#
3183        );
3184    }
3185
3186    #[test]
3187    fn rfc6532() {
3188        let message = concat!(
3189            "From: Keith Moore <moore@cs.utk.edu>\n",
3190            "To: Keld Jørn Simonsen <keld@dkuug.dk>\n",
3191            "CC: André Pirard <PIRARD@vm1.ulg.ac.be>\n",
3192            "Subject: Hello André\n",
3193            "\n\n"
3194        );
3195        let msg = MimePart::parse(message).unwrap();
3196        let list = match msg.headers().from() {
3197            Err(err) => panic!("Doh.\n{err:#}"),
3198            Ok(list) => list,
3199        };
3200        k9::snapshot!(
3201            list,
3202            r#"
3203Some(
3204    MailboxList(
3205        [
3206            Mailbox {
3207                name: Some(
3208                    "Keith Moore",
3209                ),
3210                address: AddrSpec {
3211                    local_part: "moore",
3212                    domain: "cs.utk.edu",
3213                },
3214            },
3215        ],
3216    ),
3217)
3218"#
3219        );
3220
3221        let list = match msg.headers().to() {
3222            Err(err) => panic!("Doh.\n{err:#}"),
3223            Ok(list) => list,
3224        };
3225        k9::snapshot!(
3226            list,
3227            r#"
3228Some(
3229    AddressList(
3230        [
3231            Mailbox(
3232                Mailbox {
3233                    name: Some(
3234                        "Keld Jørn Simonsen",
3235                    ),
3236                    address: AddrSpec {
3237                        local_part: "keld",
3238                        domain: "dkuug.dk",
3239                    },
3240                },
3241            ),
3242        ],
3243    ),
3244)
3245"#
3246        );
3247
3248        let list = match msg.headers().cc() {
3249            Err(err) => panic!("Doh.\n{err:#}"),
3250            Ok(list) => list,
3251        };
3252        k9::snapshot!(
3253            list,
3254            r#"
3255Some(
3256    AddressList(
3257        [
3258            Mailbox(
3259                Mailbox {
3260                    name: Some(
3261                        "André Pirard",
3262                    ),
3263                    address: AddrSpec {
3264                        local_part: "PIRARD",
3265                        domain: "vm1.ulg.ac.be",
3266                    },
3267                },
3268            ),
3269        ],
3270    ),
3271)
3272"#
3273        );
3274        let list = match msg.headers().subject() {
3275            Err(err) => panic!("Doh.\n{err:#}"),
3276            Ok(list) => list,
3277        };
3278        k9::snapshot!(
3279            list,
3280            r#"
3281Some(
3282    "Hello André",
3283)
3284"#
3285        );
3286    }
3287
3288    #[test]
3289    fn unstructured_bare_non_ascii() {
3290        // Direct test of unstructured header parsing with bare UTF-8
3291        // (no encoded-word), exercising obs_utext -> utf8_non_ascii
3292        let message = "Subject: Héllo wörld äöü\n\n\n";
3293        let msg = MimePart::parse(message).unwrap();
3294        k9::snapshot!(
3295            msg.headers().subject().unwrap(),
3296            r#"
3297Some(
3298    "Héllo wörld äöü",
3299)
3300"#
3301        );
3302
3303        // Subject with CJK characters
3304        let message = "Subject: 件名テスト\n\n\n";
3305        let msg = MimePart::parse(message).unwrap();
3306        k9::snapshot!(
3307            msg.headers().subject().unwrap(),
3308            r#"
3309Some(
3310    "件名テスト",
3311)
3312"#
3313        );
3314    }
3315
3316    #[test]
3317    fn unstructured_raw_shift_jis() {
3318        // Raw Shift-JIS bytes in a Subject header (not wrapped in an
3319        // RFC 2047 encoded-word). "テスト" in Shift-JIS is:
3320        //   テ=0x83 0x65  ス=0x83 0x58  ト=0x83 0x67
3321        // These bytes are not valid UTF-8 (0x83 is a continuation byte
3322        // appearing as a lead byte). With utf8_non_ascii validation,
3323        // the parser will not match them as non-ASCII text.
3324        let message = b"Subject: \x83\x65\x83\x58\x83\x67\n\n\n";
3325
3326        // Structural parse succeeds: the message is split into headers
3327        // and body, and the Subject header is recognized.
3328        let msg = MimePart::parse(message.as_slice()).unwrap();
3329        let subject_header = msg.headers().get_first("Subject").unwrap();
3330        k9::assert_equal!(
3331            subject_header.get_raw_value(),
3332            b"\x83\x65\x83\x58\x83\x67".as_slice()
3333        );
3334
3335        // Semantic parse of the value as unstructured text fails because
3336        // the raw bytes are not valid UTF-8.
3337        k9::snapshot!(
3338            msg.headers().subject(),
3339            r#"
3340Err(
3341    InvalidHeaderValueDuringGet {
3342        header_name: "Subject",
3343        error: HeaderParse(
3344            "Error at line 1, in Eof:
3345\\x83e\\x83X\\x83g
3346^_____
3347
3348",
3349        ),
3350    },
3351)
3352"#
3353        );
3354    }
3355
3356    #[test]
3357    fn rfc2047_bogus() {
3358        let message = concat!(
3359            "From: =?US-OSCII?Q?Keith_Moore?= <moore@cs.utk.edu>\n",
3360            "To: =?ISO-8859-1*en-us?Q?Keld_J=F8rn_Simonsen?= <keld@dkuug.dk>\n",
3361            "CC: =?ISO-8859-1?Q?Andr=E?= Pirard <PIRARD@vm1.ulg.ac.be>\n",
3362            "Subject: Hello =?ISO-8859-1?B?SWYgeW91IGNhb!ByZWFkIHRoaXMgeW8=?=\n",
3363            "  =?ISO-8859-2?B?dSB1bmRlcnN0YW5kIHRoZSBleGFtcGxlLg==?=\n",
3364            "\n\n"
3365        );
3366        let msg = MimePart::parse(message).unwrap();
3367
3368        // Invalid charset causes encoded_word to fail and we will instead match
3369        // obs_utext and return it as it was
3370        k9::assert_equal!(
3371            msg.headers().from().unwrap().unwrap().0[0]
3372                .name
3373                .as_ref()
3374                .unwrap(),
3375            "=?US-OSCII?Q?Keith_Moore?="
3376        );
3377
3378        match &msg.headers().cc().unwrap().unwrap().0[0] {
3379            Address::Mailbox(mbox) => {
3380                // 'Andr=E9?=' is in the non-bogus example below, but above we
3381                // broke it as 'Andr=E?=', and instead of triggering a qp decode
3382                // error, it is passed through here as-is
3383                k9::assert_equal!(mbox.name.as_ref().unwrap(), "Andr=E Pirard");
3384            }
3385            wat => panic!("should not have {wat:?}"),
3386        }
3387
3388        // The invalid base64 (an I was replaced by an !) is interpreted as obs_utext
3389        // and passed through to us
3390        k9::assert_equal!(
3391            msg.headers().subject().unwrap().unwrap(),
3392            "Hello =?ISO-8859-1?B?SWYgeW91IGNhb!ByZWFkIHRoaXMgeW8=?= u understand the example."
3393        );
3394    }
3395
3396    #[test]
3397    fn attachment_filename_mess_totally_bogus() {
3398        let message = concat!("Content-Disposition: attachment; filename=@\n", "\n\n");
3399        let msg = MimePart::parse(message).unwrap();
3400        eprintln!("{msg:#?}");
3401
3402        assert!(msg
3403            .conformance()
3404            .contains(MessageConformance::INVALID_MIME_HEADERS));
3405        msg.headers().content_disposition().unwrap_err();
3406
3407        // There is no Content-Disposition in the rebuilt message, because
3408        // there was no valid Content-Disposition in what we parsed
3409        let rebuilt = msg.rebuild(None).unwrap();
3410        k9::assert_equal!(rebuilt.headers().content_disposition(), Ok(None));
3411    }
3412
3413    #[test]
3414    fn attachment_filename_mess_aberrant() {
3415        let message = concat!(
3416            "Content-Disposition: attachment; filename= =?UTF-8?B?5pel5pys6Kqe44Gu5re75LuY?=\n",
3417            "\n\n"
3418        );
3419        let msg = MimePart::parse(message).unwrap();
3420
3421        let cd = msg.headers().content_disposition().unwrap().unwrap();
3422        k9::assert_equal!(cd.get("filename").unwrap(), "日本語の添付");
3423
3424        let encoded = cd.encode_value();
3425        k9::assert_equal!(encoded, "attachment;\r\n\tfilename==?UTF-8?q?=E6=97=A5=E6=9C=AC=E8=AA=9E=E3=81=AE=E6=B7=BB=E4=BB=98?=");
3426    }
3427
3428    #[test]
3429    fn attachment_filename_mess_gmail() {
3430        let message = concat!(
3431            "Content-Disposition: attachment; filename=\"=?UTF-8?B?5pel5pys6Kqe44Gu5re75LuY?=\"\n",
3432            "Content-Type: text/plain;\n",
3433            "   name=\"=?UTF-8?B?5pel5pys6Kqe44Gu5re75LuY?=\"\n",
3434            "\n\n"
3435        );
3436        let msg = MimePart::parse(message).unwrap();
3437
3438        let cd = msg.headers().content_disposition().unwrap().unwrap();
3439        k9::assert_equal!(cd.get("filename").unwrap(), "日本語の添付");
3440        let encoded = cd.encode_value();
3441        k9::assert_equal!(encoded, "attachment;\r\n\tfilename=\"=?UTF-8?q?=E6=97=A5=E6=9C=AC=E8=AA=9E=E3=81=AE=E6=B7=BB=E4=BB=98?=\"");
3442
3443        let ct = msg.headers().content_type().unwrap().unwrap();
3444        k9::assert_equal!(ct.get("name").unwrap(), "日本語の添付");
3445    }
3446
3447    #[test]
3448    fn attachment_filename_mess_fastmail() {
3449        let message = concat!(
3450            "Content-Disposition: attachment;\n",
3451            "  filename*0*=utf-8''%E6%97%A5%E6%9C%AC%E8%AA%9E%E3%81%AE%E6%B7%BB%E4%BB%98;\n",
3452            "  filename*1*=.txt\n",
3453            "Content-Type: text/plain;\n",
3454            "   name=\"=?UTF-8?Q?=E6=97=A5=E6=9C=AC=E8=AA=9E=E3=81=AE=E6=B7=BB=E4=BB=98.txt?=\"\n",
3455            "   x-name=\"=?UTF-8?Q?=E6=97=A5=E6=9C=AC=E8=AA=9E=E3=81=AE=E6=B7=BB=E4=BB=98.txt?=bork\"\n",
3456            "\n\n"
3457        );
3458        let msg = MimePart::parse(message).unwrap();
3459
3460        let cd = msg.headers().content_disposition().unwrap().unwrap();
3461        k9::assert_equal!(cd.get("filename").unwrap(), "日本語の添付.txt");
3462
3463        let ct = msg.headers().content_type().unwrap().unwrap();
3464        eprintln!("{ct:#?}");
3465        k9::assert_equal!(ct.get("name").unwrap(), "日本語の添付.txt");
3466        k9::assert_equal!(
3467            ct.get("x-name").unwrap(),
3468            "=?UTF-8?Q?=E6=97=A5=E6=9C=AC=E8=AA=9E=E3=81=AE=E6=B7=BB=E4=BB=98.txt?=bork"
3469        );
3470    }
3471
3472    #[test]
3473    fn rfc2047() {
3474        let message = concat!(
3475            "From: =?US-ASCII?Q?Keith_Moore?= <moore@cs.utk.edu>\n",
3476            "To: =?ISO-8859-1*en-us?Q?Keld_J=F8rn_Simonsen?= <keld@dkuug.dk>\n",
3477            "CC: =?ISO-8859-1?Q?Andr=E9?= Pirard <PIRARD@vm1.ulg.ac.be>\n",
3478            "Subject: Hello =?ISO-8859-1?B?SWYgeW91IGNhbiByZWFkIHRoaXMgeW8=?=\n",
3479            "  =?ISO-8859-2?B?dSB1bmRlcnN0YW5kIHRoZSBleGFtcGxlLg==?=\n",
3480            "\n\n"
3481        );
3482        let msg = MimePart::parse(message).unwrap();
3483        let list = match msg.headers().from() {
3484            Err(err) => panic!("Doh.\n{err:#}"),
3485            Ok(list) => list,
3486        };
3487        k9::snapshot!(
3488            list,
3489            r#"
3490Some(
3491    MailboxList(
3492        [
3493            Mailbox {
3494                name: Some(
3495                    "Keith Moore",
3496                ),
3497                address: AddrSpec {
3498                    local_part: "moore",
3499                    domain: "cs.utk.edu",
3500                },
3501            },
3502        ],
3503    ),
3504)
3505"#
3506        );
3507
3508        let list = match msg.headers().to() {
3509            Err(err) => panic!("Doh.\n{err:#}"),
3510            Ok(list) => list,
3511        };
3512        k9::snapshot!(
3513            list,
3514            r#"
3515Some(
3516    AddressList(
3517        [
3518            Mailbox(
3519                Mailbox {
3520                    name: Some(
3521                        "Keld Jørn Simonsen",
3522                    ),
3523                    address: AddrSpec {
3524                        local_part: "keld",
3525                        domain: "dkuug.dk",
3526                    },
3527                },
3528            ),
3529        ],
3530    ),
3531)
3532"#
3533        );
3534
3535        let list = match msg.headers().cc() {
3536            Err(err) => panic!("Doh.\n{err:#}"),
3537            Ok(list) => list,
3538        };
3539        k9::snapshot!(
3540            list,
3541            r#"
3542Some(
3543    AddressList(
3544        [
3545            Mailbox(
3546                Mailbox {
3547                    name: Some(
3548                        "André Pirard",
3549                    ),
3550                    address: AddrSpec {
3551                        local_part: "PIRARD",
3552                        domain: "vm1.ulg.ac.be",
3553                    },
3554                },
3555            ),
3556        ],
3557    ),
3558)
3559"#
3560        );
3561        let list = match msg.headers().subject() {
3562            Err(err) => panic!("Doh.\n{err:#}"),
3563            Ok(list) => list,
3564        };
3565        k9::snapshot!(
3566            list,
3567            r#"
3568Some(
3569    "Hello If you can read this you understand the example.",
3570)
3571"#
3572        );
3573
3574        k9::snapshot!(
3575            BString::from(msg.rebuild(None).unwrap().to_message_bytes().unwrap()),
3576            r#"
3577Content-Type: text/plain;\r
3578\tcharset="us-ascii"\r
3579Content-Transfer-Encoding: quoted-printable\r
3580From: "Keith Moore" <moore@cs.utk.edu>\r
3581To: =?UTF-8?q?Keld_J=C3=B8rn_Simonsen?= <keld@dkuug.dk>\r
3582Cc: =?UTF-8?q?Andr=C3=A9_Pirard?= <PIRARD@vm1.ulg.ac.be>\r
3583Subject: Hello If you can read this you understand the example.\r
3584\r
3585=0A\r
3586
3587"#
3588        );
3589    }
3590
3591    #[test]
3592    fn group_addresses() {
3593        let message = concat!(
3594            "To: A Group:Ed Jones <c@a.test>,joe@where.test,John <jdoe@one.test>;\n",
3595            "Cc: Undisclosed recipients:;\n",
3596            "\n\n\n"
3597        );
3598        let msg = MimePart::parse(message).unwrap();
3599        let list = match msg.headers().to() {
3600            Err(err) => panic!("Doh.\n{err:#}"),
3601            Ok(list) => list.unwrap(),
3602        };
3603
3604        k9::snapshot!(
3605            list.encode_value(),
3606            r#"
3607A Group:"Ed Jones" <c@a.test>,\r
3608\t<joe@where.test>,\r
3609\tJohn <jdoe@one.test>;
3610"#
3611        );
3612
3613        let round_trip = Header::new("To", list.clone());
3614        k9::assert_equal!(list, round_trip.as_address_list().unwrap());
3615
3616        k9::snapshot!(
3617            list,
3618            r#"
3619AddressList(
3620    [
3621        Group {
3622            name: "A Group",
3623            entries: MailboxList(
3624                [
3625                    Mailbox {
3626                        name: Some(
3627                            "Ed Jones",
3628                        ),
3629                        address: AddrSpec {
3630                            local_part: "c",
3631                            domain: "a.test",
3632                        },
3633                    },
3634                    Mailbox {
3635                        name: None,
3636                        address: AddrSpec {
3637                            local_part: "joe",
3638                            domain: "where.test",
3639                        },
3640                    },
3641                    Mailbox {
3642                        name: Some(
3643                            "John",
3644                        ),
3645                        address: AddrSpec {
3646                            local_part: "jdoe",
3647                            domain: "one.test",
3648                        },
3649                    },
3650                ],
3651            ),
3652        },
3653    ],
3654)
3655"#
3656        );
3657
3658        let list = match msg.headers().cc() {
3659            Err(err) => panic!("Doh.\n{err:#}"),
3660            Ok(list) => list,
3661        };
3662        k9::snapshot!(
3663            list,
3664            r#"
3665Some(
3666    AddressList(
3667        [
3668            Group {
3669                name: "Undisclosed recipients",
3670                entries: MailboxList(
3671                    [],
3672                ),
3673            },
3674        ],
3675    ),
3676)
3677"#
3678        );
3679    }
3680
3681    #[test]
3682    fn message_id() {
3683        let message = concat!(
3684            "Message-Id: <foo@example.com>\n",
3685            "References: <a@example.com> <b@example.com>\n",
3686            "  <\"legacy\"@example.com>\n",
3687            "  <literal@[127.0.0.1]>\n",
3688            "\n\n\n"
3689        );
3690        let msg = MimePart::parse(message).unwrap();
3691        let list = match msg.headers().message_id() {
3692            Err(err) => panic!("Doh.\n{err:#}"),
3693            Ok(list) => list,
3694        };
3695        k9::snapshot!(
3696            list,
3697            r#"
3698Some(
3699    MessageID(
3700        "foo@example.com",
3701    ),
3702)
3703"#
3704        );
3705
3706        let list = match msg.headers().references() {
3707            Err(err) => panic!("Doh.\n{err:#}"),
3708            Ok(list) => list,
3709        };
3710        k9::snapshot!(
3711            list,
3712            r#"
3713Some(
3714    [
3715        MessageID(
3716            "a@example.com",
3717        ),
3718        MessageID(
3719            "b@example.com",
3720        ),
3721        MessageID(
3722            "legacy@example.com",
3723        ),
3724        MessageID(
3725            "literal@[127.0.0.1]",
3726        ),
3727    ],
3728)
3729"#
3730        );
3731    }
3732
3733    #[test]
3734    fn content_type() {
3735        let message = "Content-Type: text/plain\n\n\n\n";
3736        let msg = MimePart::parse(message).unwrap();
3737        let params = match msg.headers().content_type() {
3738            Err(err) => panic!("Doh.\n{err:#}"),
3739            Ok(params) => params,
3740        };
3741        k9::snapshot!(
3742            params,
3743            r#"
3744Some(
3745    MimeParameters {
3746        value: "text/plain",
3747        parameters: [],
3748    },
3749)
3750"#
3751        );
3752
3753        let message = "Content-Type: text/plain; charset=us-ascii\n\n\n\n";
3754        let msg = MimePart::parse(message).unwrap();
3755        let params = match msg.headers().content_type() {
3756            Err(err) => panic!("Doh.\n{err:#}"),
3757            Ok(params) => params.unwrap(),
3758        };
3759
3760        k9::snapshot!(
3761            params.get("charset"),
3762            r#"
3763Some(
3764    "us-ascii",
3765)
3766"#
3767        );
3768        k9::snapshot!(
3769            params,
3770            r#"
3771MimeParameters {
3772    value: "text/plain",
3773    parameters: [
3774        MimeParameter {
3775            name: "charset",
3776            section: None,
3777            mime_charset: None,
3778            mime_language: None,
3779            encoding: None,
3780            value: "us-ascii",
3781        },
3782    ],
3783}
3784"#
3785        );
3786
3787        let message = "Content-Type: text/plain; charset=\"us-ascii\"\n\n\n\n";
3788        let msg = MimePart::parse(message).unwrap();
3789        let params = match msg.headers().content_type() {
3790            Err(err) => panic!("Doh.\n{err:#}"),
3791            Ok(params) => params,
3792        };
3793        k9::snapshot!(
3794            params,
3795            r#"
3796Some(
3797    MimeParameters {
3798        value: "text/plain",
3799        parameters: [
3800            MimeParameter {
3801                name: "charset",
3802                section: None,
3803                mime_charset: None,
3804                mime_language: None,
3805                encoding: None,
3806                value: "us-ascii",
3807            },
3808        ],
3809    },
3810)
3811"#
3812        );
3813    }
3814
3815    #[test]
3816    fn content_type_rfc2231() {
3817        // This example is taken from the errata for rfc2231.
3818        // <https://www.rfc-editor.org/errata/eid590>
3819        let message = concat!(
3820            "Content-Type: application/x-stuff;\n",
3821            "\ttitle*0*=us-ascii'en'This%20is%20even%20more%20;\n",
3822            "\ttitle*1*=%2A%2A%2Afun%2A%2A%2A%20;\n",
3823            "\ttitle*2=\"isn't it!\"\n",
3824            "\n\n\n"
3825        );
3826        let msg = MimePart::parse(message).unwrap();
3827        let mut params = match msg.headers().content_type() {
3828            Err(err) => panic!("Doh.\n{err:#}"),
3829            Ok(params) => params.unwrap(),
3830        };
3831
3832        let original_title = params.get("title");
3833        k9::snapshot!(
3834            &original_title,
3835            r#"
3836Some(
3837    "This is even more ***fun*** isn't it!",
3838)
3839"#
3840        );
3841
3842        k9::snapshot!(
3843            &params,
3844            r#"
3845MimeParameters {
3846    value: "application/x-stuff",
3847    parameters: [
3848        MimeParameter {
3849            name: "title",
3850            section: Some(
3851                0,
3852            ),
3853            mime_charset: Some(
3854                "us-ascii",
3855            ),
3856            mime_language: Some(
3857                "en",
3858            ),
3859            encoding: Rfc2231,
3860            value: "This%20is%20even%20more%20",
3861        },
3862        MimeParameter {
3863            name: "title",
3864            section: Some(
3865                1,
3866            ),
3867            mime_charset: None,
3868            mime_language: None,
3869            encoding: Rfc2231,
3870            value: "%2A%2A%2Afun%2A%2A%2A%20",
3871        },
3872        MimeParameter {
3873            name: "title",
3874            section: Some(
3875                2,
3876            ),
3877            mime_charset: None,
3878            mime_language: None,
3879            encoding: None,
3880            value: "isn't it!",
3881        },
3882    ],
3883}
3884"#
3885        );
3886
3887        k9::snapshot!(
3888            params.encode_value(),
3889            r#"
3890application/x-stuff;\r
3891\ttitle="This is even more ***fun*** isn't it!"
3892"#
3893        );
3894
3895        params.set("foo", "bar 💩");
3896
3897        params.set(
3898            "long",
3899            "this is some text that should wrap because \
3900                it should be a good bit longer than our target maximum \
3901                length for this sort of thing, and hopefully we see at \
3902                least three lines produced as a result of setting \
3903                this value in this way",
3904        );
3905
3906        params.set(
3907            "longernnamethananyoneshouldreallyuse",
3908            "this is some text that should wrap because \
3909                it should be a good bit longer than our target maximum \
3910                length for this sort of thing, and hopefully we see at \
3911                least three lines produced as a result of setting \
3912                this value in this way",
3913        );
3914
3915        k9::snapshot!(
3916            params.encode_value(),
3917            r#"
3918application/x-stuff;\r
3919\tfoo*=UTF-8''bar%20%F0%9F%92%A9;\r
3920\tlong*0="this is some text that should wrap because it should be a good bi";\r
3921\tlong*1="t longer than our target maximum length for this sort of thing, a";\r
3922\tlong*2="nd hopefully we see at least three lines produced as a result of ";\r
3923\tlong*3="setting this value in this way";\r
3924\tlongernnamethananyoneshouldreallyuse*0="this is some text that should wra";\r
3925\tlongernnamethananyoneshouldreallyuse*1="p because it should be a good bit";\r
3926\tlongernnamethananyoneshouldreallyuse*2=" longer than our target maximum l";\r
3927\tlongernnamethananyoneshouldreallyuse*3="ength for this sort of thing, and";\r
3928\tlongernnamethananyoneshouldreallyuse*4=" hopefully we see at least three ";\r
3929\tlongernnamethananyoneshouldreallyuse*5="lines produced as a result of set";\r
3930\tlongernnamethananyoneshouldreallyuse*6="ting this value in this way";\r
3931\ttitle="This is even more ***fun*** isn't it!"
3932"#
3933        );
3934    }
3935
3936    #[test]
3937    fn content_type_long_parameter_name() {
3938        // A parameter name long enough that the fold framing exceeds the target
3939        // line width used to drive an integer underflow (issue 608). Encoding
3940        // must not panic, and each line must contain at least one character of
3941        // the value.
3942        let name = "x".repeat(70);
3943        let mut params = MimeParameters::new("text/plain");
3944        params.set(&name, "value");
3945
3946        k9::snapshot!(
3947            params.encode_value(),
3948            r#"
3949text/plain;\r
3950\txxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx*0="v";\r
3951\txxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx*1="a";\r
3952\txxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx*2="l";\r
3953\txxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx*3="u";\r
3954\txxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx*4="e"
3955"#
3956        );
3957    }
3958
3959    #[test]
3960    fn parameter_map_groups_case_insensitively() {
3961        // Names differing only in case collapse to one entry keyed by the first
3962        // spelling seen. A repeated simple parameter is malformed and the last
3963        // occurrence wins.
3964        let params =
3965            Parser::parse_content_type_header(b"text/plain; Charset=utf-8; charset=latin-1")
3966                .unwrap();
3967        let map = params.parameter_map();
3968        k9::assert_equal!(map.len(), 1);
3969        k9::assert_equal!(map.get(BStr::new("Charset")).unwrap(), "latin-1");
3970        k9::assert_equal!(params.get("CHARSET").unwrap(), "latin-1");
3971    }
3972
3973    #[test]
3974    fn parameter_map_many_distinct_parameters() {
3975        // A header with many distinct parameters decodes to one map entry per
3976        // parameter.
3977        let mut value = b"text/plain".to_vec();
3978        for i in 0..2000 {
3979            value.extend_from_slice(format!("; p{i}=v{i}").as_bytes());
3980        }
3981        let params = Parser::parse_content_type_header(&value).unwrap();
3982        let map = params.parameter_map();
3983        k9::assert_equal!(map.len(), 2000);
3984        k9::assert_equal!(map.get(BStr::new("p0")).unwrap(), "v0");
3985        k9::assert_equal!(map.get(BStr::new("p1999")).unwrap(), "v1999");
3986    }
3987
3988    #[test]
3989    fn merge_missing_parameters_skips_present_names() {
3990        let mut dest = Parser::parse_content_type_header(b"text/plain; charset=utf-8").unwrap();
3991        let incoming =
3992            Parser::parse_content_type_header(b"text/plain; CharSet=latin-1; name=file.txt")
3993                .unwrap();
3994        dest.merge_missing_parameters(incoming.parameter_map());
3995        // charset is already present (case-insensitively) and keeps its value;
3996        // name is new and is added.
3997        k9::assert_equal!(dest.get("charset").unwrap(), "utf-8");
3998        k9::assert_equal!(dest.get("name").unwrap(), "file.txt");
3999    }
4000
4001    #[test]
4002    fn duplicate_simple_parameter_last_wins() {
4003        // A repeated simple (unsectioned) parameter keeps the last value.
4004        let params =
4005            Parser::parse_content_type_header(b"text/plain; charset=utf-8; charset=latin-1")
4006                .unwrap();
4007        k9::assert_equal!(params.get("charset").unwrap(), "latin-1");
4008    }
4009
4010    #[test]
4011    fn multi_section_parameter_orders_numerically() {
4012        // Sections are compared as integers, not lexically. A lexical
4013        // comparison would place *10 and *11 between *1 and *2. The sections
4014        // are supplied out of order to prove the decode sorts them.
4015        let mut header = b"text/plain".to_vec();
4016        for section in [0u32, 10, 2, 11, 1, 3, 4, 5, 6, 7, 8, 9] {
4017            header.extend_from_slice(format!("; title*{section}=v{section}x").as_bytes());
4018        }
4019        let params = Parser::parse_content_type_header(&header).unwrap();
4020        k9::assert_equal!(
4021            params.get("title").unwrap(),
4022            "v0xv1xv2xv3xv4xv5xv6xv7xv8xv9xv10xv11x"
4023        );
4024    }
4025
4026    #[test]
4027    fn merge_missing_parameters_reencodes_non_ascii() {
4028        // A merged non-ASCII value round-trips through get, and encode_value
4029        // renders it as RFC 2231 charset-tagged continuation sections.
4030        let mut dest = MimeParameters::new("text/plain");
4031        let mut incoming = BTreeMap::new();
4032        incoming.insert(
4033            BString::from("title"),
4034            BString::from("\u{65e5}\u{672c}\u{8a9e} ".repeat(6).trim_end().as_bytes()),
4035        );
4036        dest.merge_missing_parameters(incoming);
4037        k9::assert_equal!(
4038            dest.get("title").unwrap(),
4039            "\u{65e5}\u{672c}\u{8a9e} \u{65e5}\u{672c}\u{8a9e} \u{65e5}\u{672c}\u{8a9e} \u{65e5}\u{672c}\u{8a9e} \u{65e5}\u{672c}\u{8a9e} \u{65e5}\u{672c}\u{8a9e}"
4040        );
4041        k9::snapshot!(
4042            dest.encode_value(),
4043            r#"
4044text/plain;\r
4045\ttitle*0*=UTF-8''%E6%97%A5%E6%9C%AC%E8%AA%9E%20%E6%97%A5%E6%9C%AC%E8%AA%9E%20;\r
4046\ttitle*1*=%E6%97%A5%E6%9C%AC%E8%AA%9E%20%E6%97%A5%E6%9C%AC%E8%AA%9E%20%E6%97%A5;\r
4047\ttitle*2*=%E6%9C%AC%E8%AA%9E%20%E6%97%A5%E6%9C%AC%E8%AA%9E
4048"#
4049        );
4050    }
4051
4052    /// <https://datatracker.ietf.org/doc/html/rfc8601#appendix-B.2>
4053    #[test]
4054    fn authentication_results_b_2() {
4055        let ar = Header::with_name_value("Authentication-Results", "example.org 1; none");
4056        let ar = ar.as_authentication_results().unwrap();
4057        k9::snapshot!(
4058            &ar,
4059            r#"
4060AuthenticationResults {
4061    serv_id: "example.org",
4062    version: Some(
4063        1,
4064    ),
4065    results: [],
4066}
4067"#
4068        );
4069
4070        k9::snapshot!(ar.encode_value(), "example.org 1; none");
4071    }
4072
4073    /// <https://datatracker.ietf.org/doc/html/rfc8601#appendix-B.3>
4074    #[test]
4075    fn authentication_results_b_3() {
4076        let ar = Header::with_name_value(
4077            "Authentication-Results",
4078            "example.com; spf=pass smtp.mailfrom=example.net",
4079        );
4080        k9::snapshot!(
4081            ar.as_authentication_results(),
4082            r#"
4083Ok(
4084    AuthenticationResults {
4085        serv_id: "example.com",
4086        version: None,
4087        results: [
4088            AuthenticationResult {
4089                method: "spf",
4090                method_version: None,
4091                result: "pass",
4092                reason: None,
4093                props: {
4094                    "smtp.mailfrom": "example.net",
4095                },
4096            },
4097        ],
4098    },
4099)
4100"#
4101        );
4102    }
4103
4104    /// <https://datatracker.ietf.org/doc/html/rfc8601#appendix-B.4>
4105    #[test]
4106    fn authentication_results_b_4() {
4107        let ar = Header::with_name_value(
4108            "Authentication-Results",
4109            concat!(
4110                "example.com;\n",
4111                "\tauth=pass (cram-md5) smtp.auth=sender@example.net;\n",
4112                "\tspf=pass smtp.mailfrom=example.net"
4113            ),
4114        );
4115        k9::snapshot!(
4116            ar.as_authentication_results(),
4117            r#"
4118Ok(
4119    AuthenticationResults {
4120        serv_id: "example.com",
4121        version: None,
4122        results: [
4123            AuthenticationResult {
4124                method: "auth",
4125                method_version: None,
4126                result: "pass",
4127                reason: None,
4128                props: {
4129                    "smtp.auth": "sender@example.net",
4130                },
4131            },
4132            AuthenticationResult {
4133                method: "spf",
4134                method_version: None,
4135                result: "pass",
4136                reason: None,
4137                props: {
4138                    "smtp.mailfrom": "example.net",
4139                },
4140            },
4141        ],
4142    },
4143)
4144"#
4145        );
4146
4147        let ar = Header::with_name_value(
4148            "Authentication-Results",
4149            "example.com; iprev=pass\n\tpolicy.iprev=192.0.2.200",
4150        );
4151        k9::snapshot!(
4152            ar.as_authentication_results(),
4153            r#"
4154Ok(
4155    AuthenticationResults {
4156        serv_id: "example.com",
4157        version: None,
4158        results: [
4159            AuthenticationResult {
4160                method: "iprev",
4161                method_version: None,
4162                result: "pass",
4163                reason: None,
4164                props: {
4165                    "policy.iprev": "192.0.2.200",
4166                },
4167            },
4168        ],
4169    },
4170)
4171"#
4172        );
4173    }
4174
4175    /// <https://datatracker.ietf.org/doc/html/rfc8601#appendix-B.5>
4176    #[test]
4177    fn authentication_results_b_5() {
4178        let ar = Header::with_name_value(
4179            "Authentication-Results",
4180            "example.com;\n\tdkim=pass (good signature) header.d=example.com",
4181        );
4182        k9::snapshot!(
4183            ar.as_authentication_results(),
4184            r#"
4185Ok(
4186    AuthenticationResults {
4187        serv_id: "example.com",
4188        version: None,
4189        results: [
4190            AuthenticationResult {
4191                method: "dkim",
4192                method_version: None,
4193                result: "pass",
4194                reason: None,
4195                props: {
4196                    "header.d": "example.com",
4197                },
4198            },
4199        ],
4200    },
4201)
4202"#
4203        );
4204
4205        let ar = Header::with_name_value(
4206            "Authentication-Results",
4207            "example.com;\n\tauth=pass (cram-md5) smtp.auth=sender@example.com;\n\tspf=fail smtp.mailfrom=example.com"
4208        );
4209        let ar = ar.as_authentication_results().unwrap();
4210        k9::snapshot!(
4211            &ar,
4212            r#"
4213AuthenticationResults {
4214    serv_id: "example.com",
4215    version: None,
4216    results: [
4217        AuthenticationResult {
4218            method: "auth",
4219            method_version: None,
4220            result: "pass",
4221            reason: None,
4222            props: {
4223                "smtp.auth": "sender@example.com",
4224            },
4225        },
4226        AuthenticationResult {
4227            method: "spf",
4228            method_version: None,
4229            result: "fail",
4230            reason: None,
4231            props: {
4232                "smtp.mailfrom": "example.com",
4233            },
4234        },
4235    ],
4236}
4237"#
4238        );
4239
4240        k9::snapshot!(
4241            ar.encode_value(),
4242            r#"
4243example.com;\r
4244\tauth=pass\r
4245\tsmtp.auth=sender@example.com;\r
4246\tspf=fail\r
4247\tsmtp.mailfrom=example.com
4248"#
4249        );
4250    }
4251
4252    /// <https://datatracker.ietf.org/doc/html/rfc8601#appendix-B.6>
4253    #[test]
4254    fn authentication_results_b_6() {
4255        let ar = Header::with_name_value(
4256            "Authentication-Results",
4257            concat!(
4258                "example.com;\n",
4259                "\tdkim=pass reason=\"good signature\"\n",
4260                "\theader.i=@mail-router.example.net;\n",
4261                "\tdkim=fail reason=\"bad signature\"\n",
4262                "\theader.i=@newyork.example.com"
4263            ),
4264        );
4265        let ar = match ar.as_authentication_results() {
4266            Err(err) => panic!("\n{err}"),
4267            Ok(ar) => ar,
4268        };
4269
4270        k9::snapshot!(
4271            &ar,
4272            r#"
4273AuthenticationResults {
4274    serv_id: "example.com",
4275    version: None,
4276    results: [
4277        AuthenticationResult {
4278            method: "dkim",
4279            method_version: None,
4280            result: "pass",
4281            reason: Some(
4282                "good signature",
4283            ),
4284            props: {
4285                "header.i": "@mail-router.example.net",
4286            },
4287        },
4288        AuthenticationResult {
4289            method: "dkim",
4290            method_version: None,
4291            result: "fail",
4292            reason: Some(
4293                "bad signature",
4294            ),
4295            props: {
4296                "header.i": "@newyork.example.com",
4297            },
4298        },
4299    ],
4300}
4301"#
4302        );
4303
4304        k9::snapshot!(
4305            ar.encode_value(),
4306            r#"
4307example.com;\r
4308\tdkim=pass reason="good signature"\r
4309\theader.i=@mail-router.example.net;\r
4310\tdkim=fail reason="bad signature"\r
4311\theader.i=@newyork.example.com
4312"#
4313        );
4314
4315        let ar = Header::with_name_value(
4316            "Authentication-Results",
4317            concat!(
4318                "example.net;\n",
4319                "\tdkim=pass (good signature) header.i=@newyork.example.com"
4320            ),
4321        );
4322        let ar = match ar.as_authentication_results() {
4323            Err(err) => panic!("\n{err}"),
4324            Ok(ar) => ar,
4325        };
4326
4327        k9::snapshot!(
4328            &ar,
4329            r#"
4330AuthenticationResults {
4331    serv_id: "example.net",
4332    version: None,
4333    results: [
4334        AuthenticationResult {
4335            method: "dkim",
4336            method_version: None,
4337            result: "pass",
4338            reason: None,
4339            props: {
4340                "header.i": "@newyork.example.com",
4341            },
4342        },
4343    ],
4344}
4345"#
4346        );
4347
4348        k9::snapshot!(
4349            ar.encode_value(),
4350            r#"
4351example.net;\r
4352\tdkim=pass\r
4353\theader.i=@newyork.example.com
4354"#
4355        );
4356    }
4357
4358    /// <https://datatracker.ietf.org/doc/html/rfc8601#appendix-B.7>
4359    #[test]
4360    fn authentication_results_b_7() {
4361        let ar = Header::with_name_value(
4362            "Authentication-Results",
4363            concat!(
4364                "foo.example.net (foobar) 1 (baz);\n",
4365                "\tdkim (Because I like it) / 1 (One yay) = (wait for it) fail\n",
4366                "\tpolicy (A dot can go here) . (like that) expired\n",
4367                "\t(this surprised me) = (as I wasn't expecting it) 1362471462"
4368            ),
4369        );
4370        let ar = match ar.as_authentication_results() {
4371            Err(err) => panic!("\n{err}"),
4372            Ok(ar) => ar,
4373        };
4374
4375        k9::snapshot!(
4376            &ar,
4377            r#"
4378AuthenticationResults {
4379    serv_id: "foo.example.net",
4380    version: Some(
4381        1,
4382    ),
4383    results: [
4384        AuthenticationResult {
4385            method: "dkim",
4386            method_version: Some(
4387                1,
4388            ),
4389            result: "fail",
4390            reason: None,
4391            props: {
4392                "policy.expired": "1362471462",
4393            },
4394        },
4395    ],
4396}
4397"#
4398        );
4399
4400        k9::snapshot!(
4401            ar.encode_value(),
4402            r#"
4403foo.example.net 1;\r
4404\tdkim/1=fail\r
4405\tpolicy.expired=1362471462
4406"#
4407        );
4408    }
4409
4410    #[test]
4411    fn arc_authentication_results_1() {
4412        let ar = Header::with_name_value(
4413            "ARC-Authentication-Results",
4414            "i=3; clochette.example.org; spf=fail
4415    smtp.from=jqd@d1.example; dkim=fail (512-bit key)
4416    header.i=@d1.example; dmarc=fail; arc=pass (as.2.gmail.example=pass,
4417    ams.2.gmail.example=pass, as.1.lists.example.org=pass,
4418    ams.1.lists.example.org=fail (message has been altered))",
4419        );
4420        let ar = match ar.as_arc_authentication_results() {
4421            Err(err) => panic!("\n{err}"),
4422            Ok(ar) => ar,
4423        };
4424
4425        k9::snapshot!(
4426            &ar,
4427            r#"
4428ARCAuthenticationResults {
4429    instance: 3,
4430    serv_id: "clochette.example.org",
4431    version: None,
4432    results: [
4433        AuthenticationResult {
4434            method: "spf",
4435            method_version: None,
4436            result: "fail",
4437            reason: None,
4438            props: {
4439                "smtp.from": "jqd@d1.example",
4440            },
4441        },
4442        AuthenticationResult {
4443            method: "dkim",
4444            method_version: None,
4445            result: "fail",
4446            reason: None,
4447            props: {
4448                "header.i": "@d1.example",
4449            },
4450        },
4451        AuthenticationResult {
4452            method: "dmarc",
4453            method_version: None,
4454            result: "fail",
4455            reason: None,
4456            props: {},
4457        },
4458        AuthenticationResult {
4459            method: "arc",
4460            method_version: None,
4461            result: "pass",
4462            reason: None,
4463            props: {},
4464        },
4465    ],
4466}
4467"#
4468        );
4469    }
4470
4471    #[test]
4472    fn bstring_utf8_serializes_utf8_as_string() {
4473        // A MessageID with pure ASCII content serializes as a JSON string
4474        let mid = MessageID(BString::from("abc123@example.com"));
4475        let json = serde_json::to_string(&mid).unwrap();
4476        k9::assert_equal!(json, r#""abc123@example.com""#);
4477    }
4478
4479    #[test]
4480    fn bstring_utf8_serializes_non_utf8_as_array() {
4481        // A MessageID with invalid UTF-8 falls back to byte array
4482        let mid = MessageID(BString::from(&b"hello\x80world"[..]));
4483        let json = serde_json::to_string(&mid).unwrap();
4484        k9::assert_equal!(json, "[104,101,108,108,111,128,119,111,114,108,100]");
4485    }
4486
4487    #[test]
4488    fn bstring_utf8_round_trip_utf8() {
4489        let mid = MessageID(BString::from("test@example.com"));
4490        let json = serde_json::to_string(&mid).unwrap();
4491        let restored: MessageID = serde_json::from_str(&json).unwrap();
4492        k9::assert_equal!(restored, mid);
4493    }
4494
4495    #[test]
4496    fn bstring_utf8_round_trip_non_utf8() {
4497        let mid = MessageID(BString::from(&b"\xff\xfe"[..]));
4498        let json = serde_json::to_string(&mid).unwrap();
4499        let restored: MessageID = serde_json::from_str(&json).unwrap();
4500        k9::assert_equal!(restored, mid);
4501    }
4502
4503    #[test]
4504    fn authentication_results_serialize_as_strings() {
4505        let ar = AuthenticationResults {
4506            serv_id: BString::from("example.com"),
4507            version: None,
4508            results: vec![AuthenticationResult {
4509                method: "dkim".into(),
4510                method_version: None,
4511                result: "pass".into(),
4512                reason: Some(BString::from("good signature")),
4513                props: BTreeMap::from([
4514                    ("header.d".into(), BString::from("example.com")),
4515                    ("header.s".into(), BString::from("selector1")),
4516                ]),
4517            }],
4518        };
4519        let json = serde_json::to_string_pretty(&ar).unwrap();
4520        // All BString fields that are valid UTF-8 should appear as JSON strings
4521        k9::assert_equal!(
4522            json,
4523            r#"{
4524  "serv_id": "example.com",
4525  "version": null,
4526  "results": [
4527    {
4528      "method": "dkim",
4529      "method_version": null,
4530      "result": "pass",
4531      "reason": "good signature",
4532      "props": {
4533        "header.d": "example.com",
4534        "header.s": "selector1"
4535      }
4536    }
4537  ]
4538}"#
4539        );
4540    }
4541
4542    #[test]
4543    fn authentication_results_round_trip() {
4544        let ar = AuthenticationResults {
4545            serv_id: BString::from("mx.example.org"),
4546            version: Some(1),
4547            results: vec![AuthenticationResult {
4548                method: "spf".into(),
4549                method_version: None,
4550                result: "pass".into(),
4551                reason: None,
4552                props: BTreeMap::from([(
4553                    "smtp.mailfrom".into(),
4554                    BString::from("sender@example.com"),
4555                )]),
4556            }],
4557        };
4558        let json = serde_json::to_string(&ar).unwrap();
4559        let restored: AuthenticationResults = serde_json::from_str(&json).unwrap();
4560        k9::assert_equal!(restored, ar);
4561    }
4562
4563    #[test]
4564    fn authentication_result_non_utf8_reason() {
4565        let ar = AuthenticationResult {
4566            method: "dkim".into(),
4567            method_version: None,
4568            result: "temperror".into(),
4569            reason: Some(BString::from(&b"bad\x80data"[..])),
4570            props: BTreeMap::new(),
4571        };
4572        let json = serde_json::to_string(&ar).unwrap();
4573        // reason should be a byte array since it contains invalid UTF-8
4574        assert!(json.contains(r#""reason":[98,97,100,128,100,97,116,97]"#));
4575        let restored: AuthenticationResult = serde_json::from_str(&json).unwrap();
4576        k9::assert_equal!(restored, ar);
4577    }
4578
4579    #[test]
4580    fn authentication_results_encode_value_with_binary() {
4581        // Construct AuthenticationResults with non-UTF-8 bytes in BString fields
4582        // and capture the encode_value() output for use in a Lua test.
4583        let ar = AuthenticationResults {
4584            serv_id: BString::from(&b"mx.ex\x80mple.com"[..]),
4585            version: None,
4586            results: vec![AuthenticationResult {
4587                method: "spf".into(),
4588                method_version: None,
4589                result: "pass".into(),
4590                reason: Some(BString::from(&b"good\xffsig"[..])),
4591                props: BTreeMap::from([(
4592                    "smtp.mailfrom".into(),
4593                    BString::from(&b"user@\xfehost"[..]),
4594                )]),
4595            }],
4596        };
4597        let encoded = ar.encode_value();
4598        k9::snapshot!(
4599            encoded,
4600            r#"
4601"mx.ex\x80mple.com";\r
4602\tspf=pass reason="good\xffsig"\r
4603\tsmtp.mailfrom="user@\xfehost"
4604"#
4605        );
4606    }
4607
4608    #[test]
4609    fn authentication_results_serv_id_quoting() {
4610        // A serv_id containing characters that need quoting is properly quoted
4611        let ar = AuthenticationResults {
4612            serv_id: BString::from("mx example.com"),
4613            version: None,
4614            results: vec![],
4615        };
4616        let encoded = ar.encode_value();
4617        k9::snapshot!(encoded, r#""mx example.com"; none"#);
4618
4619        // Normal domain-like serv_id is emitted bare
4620        let ar2 = AuthenticationResults {
4621            serv_id: BString::from("mx.example.com"),
4622            version: Some(1),
4623            results: vec![],
4624        };
4625        let encoded2 = ar2.encode_value();
4626        k9::snapshot!(&encoded2, "mx.example.com 1; none");
4627        // Bare serv_id roundtrips
4628        let parsed = Parser::parse_authentication_results_header(encoded2.as_bytes()).unwrap();
4629        k9::assert_equal!(parsed.serv_id, ar2.serv_id);
4630        k9::assert_equal!(parsed.version, Some(1));
4631    }
4632
4633    #[test]
4634    fn authentication_results_encode_drops_injected_control_chars() {
4635        // Sender-influenced values (here a DMARC policy prop and a reason)
4636        // containing CR/LF must not split the emitted header.
4637        let mut props = std::collections::BTreeMap::new();
4638        props.insert(
4639            "policy.rua".to_string(),
4640            BString::from(&b"a\r\nX-Injected: y"[..]),
4641        );
4642        let ar = AuthenticationResults {
4643            serv_id: BString::from(&b"mx.ex\r\nX-Serv: z.com"[..]),
4644            version: None,
4645            results: vec![AuthenticationResult {
4646                method: "dmarc".into(),
4647                method_version: None,
4648                result: "pass".into(),
4649                reason: Some(BString::from(&b"ok\r\nX-Evil: yes"[..])),
4650                props,
4651            }],
4652        };
4653        let encoded = ar.encode_value().to_string();
4654
4655        // The injected content is preserved minus its control characters. The
4656        // only CRLFs left are the structural folds this encoder inserts. A new
4657        // header line cannot appear beneath ours.
4658        k9::assert_equal!(
4659            encoded,
4660            "\"mx.exX-Serv: z.com\";\r\n\tdmarc=pass reason=\"okX-Evil: yes\"\
4661             \r\n\tpolicy.rua=\"aX-Injected: y\""
4662        );
4663
4664        // After removing the structural folds, nothing survives that a header
4665        // parser would treat as a line break.
4666        let unfolded = encoded.replace("\r\n\t", "");
4667        assert!(!unfolded.contains('\r'), "residual CR in {encoded:?}");
4668        assert!(!unfolded.contains('\n'), "residual LF in {encoded:?}");
4669    }
4670
4671    #[test]
4672    fn authentication_results_encode_drops_controls_in_keys_and_arc() {
4673        // A property key sourced from Lua policy can contain structural bytes;
4674        // they must not survive into the header.
4675        let mut props = std::collections::BTreeMap::new();
4676        props.insert("policy; x=evil".to_string(), BString::from("v"));
4677        let arc = ARCAuthenticationResults {
4678            instance: 1,
4679            serv_id: BString::from(&b"mx\x00.ex"[..]),
4680            version: None,
4681            results: vec![AuthenticationResult {
4682                method: "dmarc".into(),
4683                method_version: None,
4684                result: "pass".into(),
4685                reason: None,
4686                props,
4687            }],
4688        };
4689        let encoded = arc.encode_value().to_string();
4690        // The key is reduced to its mime-token characters. The NUL in the
4691        // serv_id is dropped.
4692        k9::assert_equal!(
4693            encoded,
4694            "i=1; \"mx.ex\";\r\n\tdmarc=pass\r\n\tpolicyxevil=v"
4695        );
4696    }
4697
4698    #[test]
4699    fn authentication_results_encode_omits_prop_with_empty_key() {
4700        // A key with no valid characters sanitizes to nothing. The whole prop
4701        // is dropped rather than emitting a keyless `=value`.
4702        let mut props = std::collections::BTreeMap::new();
4703        props.insert(";;".to_string(), BString::from("v"));
4704        let ar = AuthenticationResults {
4705            serv_id: BString::from("mx.example.com"),
4706            version: None,
4707            results: vec![AuthenticationResult {
4708                method: "dmarc".into(),
4709                method_version: None,
4710                result: "pass".into(),
4711                reason: None,
4712                props,
4713            }],
4714        };
4715        k9::assert_equal!(
4716            ar.encode_value().to_string(),
4717            "mx.example.com;\r\n\tdmarc=pass"
4718        );
4719    }
4720
4721    #[test]
4722    fn authentication_results_encode_preserves_unicode() {
4723        // U+010D (\u{10d}) has low byte 0x0D (CR). A byte-truncating control
4724        // check would drop it. It must survive byte-for-byte.
4725        let ar = AuthenticationResults {
4726            serv_id: BString::from("m\u{10d}.example.com"),
4727            version: None,
4728            results: vec![],
4729        };
4730        let encoded = ar.encode_value();
4731        k9::assert_equal!(encoded, "\"m\u{10d}.example.com\"; none");
4732    }
4733
4734    #[test]
4735    fn authentication_results_encode_drops_obs_qp_control_from_parsed_header() {
4736        // The parser accepts obs-qp escapes of CR/LF/NUL inside quoted
4737        // strings and stores the literal control byte in the parsed value.
4738        // Re-encoding that value (as ARC sealing does) must not emit the
4739        // control character.
4740        let header = b"\"mx.ex\\\rX-Serv: z.com\"; none";
4741        let parsed = Parser::parse_authentication_results_header(header).unwrap();
4742        assert!(
4743            parsed.serv_id.as_bytes().contains(&b'\r'),
4744            "parser should retain the raw CR"
4745        );
4746        let encoded = parsed.encode_value().to_string();
4747        let unfolded = encoded.replace("\r\n\t", "");
4748        assert!(!unfolded.contains('\r'), "residual CR in {encoded:?}");
4749        assert!(!unfolded.contains('\n'), "residual LF in {encoded:?}");
4750    }
4751
4752    #[test]
4753    fn arc_authentication_results_serialize_as_strings() {
4754        let arc = ARCAuthenticationResults {
4755            instance: 1,
4756            serv_id: BString::from("mx.example.com"),
4757            version: None,
4758            results: vec![],
4759        };
4760        let json = serde_json::to_string(&arc).unwrap();
4761        k9::assert_equal!(
4762            json,
4763            r#"{"instance":1,"serv_id":"mx.example.com","version":null,"results":[]}"#
4764        );
4765    }
4766}