fix: improve RFC 2047 decode & fix trim space

This commit is contained in:
killerprojecte
2026-06-19 22:49:20 +08:00
parent d312076d4b
commit 5604b59501
+30 -14
View File
@@ -15,6 +15,7 @@ import (
"net/textproto" "net/textproto"
"os" "os"
"path/filepath" "path/filepath"
"regexp"
"strings" "strings"
"time" "time"
@@ -476,39 +477,54 @@ func partFilename(header textproto.MIMEHeader) string {
} }
func firstAddressParts(value string) (string, string) { func firstAddressParts(value string) (string, string) {
// Proactively decode RFC 2047 encoded words before parsing, so that items, err := netmail.ParseAddressList(value)
// non-UTF-8 charsets (e.g. GBK, Shift_JIS) are handled by our
// CharsetReader instead of Go's default WordDecoder which only
// supports UTF-8 and ISO-8859-1.
decoded := decodeMIMEHeader(value)
items, err := netmail.ParseAddressList(decoded)
if err != nil || len(items) == 0 { if err != nil || len(items) == 0 {
// Still unparseable: return the decoded value and try to extract // ParseAddressList failed — attempt RFC 2047 decode on the raw header,
// a display name from the decoded string. // then retry parsing. This handles non-standard From headers where
email, name := splitNameAndEmail(decoded) // encoded words (e.g. =?UTF-8?B?…?=) cause the initial parse to fail.
return normalizeEmail(email), strings.TrimSpace(name) decoded := decodeMIMEHeader(value)
items, err = netmail.ParseAddressList(decoded)
if err != nil || len(items) == 0 {
// Still unparseable: return the decoded value and try to extract
// a display name from the decoded string.
email, name := splitNameAndEmail(decoded)
return normalizeEmail(email), strings.TrimSpace(name)
}
} }
item := items[0] item := items[0]
return normalizeEmail(item.Address), strings.TrimSpace(item.Name) // Decode item.Name individually so that non-UTF-8 charsets (e.g. GBK,
// Shift_JIS) are handled by our CharsetReader, while the address list
// structure is parsed from the raw header (avoiding commas/semicolons
// inside decoded display names breaking the parser).
return normalizeEmail(item.Address), strings.TrimSpace(decodeMIMEHeader(item.Name))
} }
// decodeMIMEHeader decodes all RFC 2047 encoded words (=?charset?encoding?data?=) // decodeMIMEHeader decodes all RFC 2047 encoded words (=?charset?encoding?data?=)
// in the given header value. Falls back to the original value on any error. // in the given header value. Falls back to the original value on any error.
// Supports non-UTF-8 charsets (e.g. GBK, GB2312, Shift_JIS) via x/text. // Supports non-UTF-8 charsets (e.g. GBK, GB2312, Shift_JIS) via x/text.
// Per RFC 2047 §6.2, linear whitespace between adjacent encoded words is
// stripped before decoding.
func decodeMIMEHeader(value string) string { func decodeMIMEHeader(value string) string {
if !strings.Contains(value, "=?") { if !strings.Contains(value, "=?") {
return value return value
} }
// RFC 2047 §6.2: ignore whitespace between adjacent encoded words.
collapsed := adjacentEncodedWordSpaceRe.ReplaceAllString(value, "$1$2")
decoder := &mime.WordDecoder{ decoder := &mime.WordDecoder{
CharsetReader: charsetReader, CharsetReader: charsetReader,
} }
decoded, err := decoder.DecodeHeader(value) decoded, err := decoder.DecodeHeader(collapsed)
if err != nil { if err != nil {
return value return value
} }
return decoded return decoded
} }
// adjacentEncodedWordSpaceRe matches whitespace between two adjacent RFC 2047
// encoded words. Per RFC 2047 §6.2, this whitespace must be ignored when
// displaying the header.
var adjacentEncodedWordSpaceRe = regexp.MustCompile(`(\?=)\s+(=\?)`)
// charsetReader converts a non-UTF-8 charset stream into UTF-8 using x/text encodings. // charsetReader converts a non-UTF-8 charset stream into UTF-8 using x/text encodings.
func charsetReader(charset string, input io.Reader) (io.Reader, error) { func charsetReader(charset string, input io.Reader) (io.Reader, error) {
charset = strings.ToLower(strings.TrimSpace(charset)) charset = strings.ToLower(strings.TrimSpace(charset))
@@ -519,7 +535,7 @@ func charsetReader(charset string, input io.Reader) (io.Reader, error) {
if err != nil { if err != nil {
return nil, fmt.Errorf("unsupported charset %q: %w", charset, err) return nil, fmt.Errorf("unsupported charset %q: %w", charset, err)
} }
if enc == encoding.Nop || enc == encoding.Replacement { if enc == nil || enc == encoding.Nop || enc == encoding.Replacement {
return nil, fmt.Errorf("unsupported charset %q", charset) return nil, fmt.Errorf("unsupported charset %q", charset)
} }
return enc.NewDecoder().Reader(input), nil return enc.NewDecoder().Reader(input), nil
@@ -534,7 +550,7 @@ func splitNameAndEmail(value string) (string, string) {
} }
// Try "Name <email>" pattern // Try "Name <email>" pattern
if idx := strings.LastIndex(value, "<"); idx >= 0 { if idx := strings.LastIndex(value, "<"); idx >= 0 {
email := strings.TrimRight(value[idx+1:], ">") email := strings.TrimSpace(strings.Trim(value[idx+1:], "> "))
name := strings.TrimSpace(strings.Trim(value[:idx], `" `)) name := strings.TrimSpace(strings.Trim(value[:idx], `" `))
if strings.Contains(email, "@") { if strings.Contains(email, "@") {
return email, name return email, name