11 "golang.org/x/text/unicode/norm"
13 "github.com/mjl-/mox/dns"
14 "github.com/mjl-/mox/mox-"
15 "github.com/mjl-/mox/smtp"
18// Parser holds the original string and string with ascii a-z upper-cased for easy
19// case-insensitive parsing.
23 o int // Offset into orig/upper.
24 smtputf8 bool // Whether SMTPUTF8 extension is enabled, making IDNA domains and utf8 localparts valid.
26 utf8LocalpartCode int // If non-zero, error for utf-8 localpart when smtputf8 not enabled.
29// toUpper upper cases bytes that are a-z. strings.ToUpper does too much. and
30// would replace invalid bytes with unicode replacement characters, which would
31// break our requirement that offsets into the original and upper case strings
32// point to the same character.
33func toUpper(s string) string {
36 if c >= 'a' && c <= 'z' {
43func newParser(s string, smtputf8 bool, conn *conn) *parser {
44 return &parser{orig: s, upper: toUpper(s), smtputf8: smtputf8, conn: conn}
47func (p *parser) xerrorf(format string, args ...any) {
48 // For submission, send the remaining unparsed line. Otherwise, only log it.
50 errmsg := "bad syntax: " + fmt.Sprintf(format, args...)
51 remaining := fmt.Sprintf(" (remaining %q)", p.orig[p.o:])
52 if p.conn.account != nil {
54 err = errors.New(errmsg)
56 err = errors.New(errmsg + remaining)
60 panic(smtpError{smtp.C501BadParamSyntax, smtp.SeProto5Syntax2, errmsg, err, false, true})
63func (p *parser) xutf8localparterrorf() {
64 code := p.utf8LocalpartCode
66 code = smtp.C550MailboxUnavail
69 xsmtpUserErrorf(code, smtp.SeMsg6NonASCIIAddrNotPermitted7, "non-ascii address not permitted without smtputf8")
72func (p *parser) empty() bool {
73 return p.o == len(p.orig)
76// note: use xend() for check for end of line with remaining white space, to be used by commands.
77func (p *parser) xempty() {
78 if p.o != len(p.orig) {
79 p.xerrorf("expected end of line")
83// check we are at the end of a command.
84func (p *parser) xend() {
85 // For submission, we are strict.
86 if p.conn.submission {
91 for _, c := range rem {
92 if c != ' ' && c != '\t' {
93 p.xerrorf("trailing data, not white space: %q", rem)
98func (p *parser) hasPrefix(s string) bool {
99 return strings.HasPrefix(p.upper[p.o:], s)
102func (p *parser) take(s string) bool {
110func (p *parser) xtake(s string) {
112 p.xerrorf("expected %q", s)
116func (p *parser) space() bool {
120func (p *parser) xspace() {
124func (p *parser) xtaken(n int) string {
125 r := p.orig[p.o : p.o+n]
130func (p *parser) remainder() string {
136func (p *parser) peekchar() rune {
137 for _, c := range p.upper[p.o:] {
143func (p *parser) xtakefn1(what string, fn func(c rune, i int) bool) string {
145 p.xerrorf("need at least one char for %s", what)
147 for i, c := range p.upper[p.o:] {
150 p.xerrorf("expected at least one char for %s", what)
158func (p *parser) xtakefn1case(what string, fn func(c rune, i int) bool) string {
160 p.xerrorf("need at least one char for %s", what)
162 for i, c := range p.orig[p.o:] {
165 p.xerrorf("expected at least one char for %s", what)
173func (p *parser) xtakefn(fn func(c rune, i int) bool) string {
174 for i, c := range p.upper[p.o:] {
182// xrawReversePath returns the raw string between the <>'s. We cannot parse it
183// immediately, because if this is an IDNA (internationalization) address, we would
184// only see the SMTPUTF8 indicator after having parsed the reverse path here. So we
185// parse the raw data here, and validate it after having seen all parameters.
187func (p *parser) xrawReversePath() string {
189 s := p.xtakefn(func(c rune, i int) bool {
196// xbareReversePath parses a reverse-path without <>, as returned by
197// xrawReversePath. It takes smtputf8 into account.
199func (p *parser) xbareReversePath() smtp.Path {
204 p.utf8LocalpartCode = smtp.C550MailboxUnavail
206 p.utf8LocalpartCode = 0
211func (p *parser) xforwardPath() smtp.Path {
213 p.utf8LocalpartCode = smtp.C553BadMailbox
215 p.utf8LocalpartCode = 0
221func (p *parser) xpath() smtp.Path {
228 p.xerrorf("path longer than 256 octets")
233func (p *parser) xbarePath() smtp.Path {
234 // We parse but ignore any source routing.
248func (p *parser) xdomain() dns.Domain {
251 s += "." + p.xsubdomain()
253 d, err := dns.ParseDomain(s)
255 p.xerrorf("parsing domain name %q: %s", s, err)
259 p.xerrorf("domain longer than 255 octets")
265func (p *parser) xsubdomain() string {
266 return p.xtakefn1("subdomain", func(c rune, i int) bool {
267 return c >= '0' && c <= '9' || c >= 'A' && c <= 'Z' || i > 0 && c == '-' || c > 0x7f && p.smtputf8
272func (p *parser) xmailbox() smtp.Path {
273 localpart := p.xlocalpart()
275 return smtp.Path{Localpart: localpart, IPDomain: p.xipdomain(false)}
279func (p *parser) xldhstr() string {
280 s := p.xtakefn1("ldh-str", func(c rune, i int) bool {
281 return c >= 'A' && c <= 'Z' || c >= '0' && c <= '9' || c == '-'
284 p.xerrorf("empty ldh-str")
285 } else if strings.HasSuffix(s, "-") {
292// parse address-literal or domain.
293func (p *parser) xipdomain(isehlo bool) dns.IPDomain {
299 if !(c >= '0' && c <= '9') {
300 addrlit := p.xldhstr()
302 if !strings.EqualFold(addrlit, "IPv6") {
303 p.xerrorf("unrecognized address literal %q", addrlit)
307 ipaddr := p.xtakefn1("address literal", func(c rune, i int) bool {
311 ip := net.ParseIP(ipaddr)
313 p.xerrorf("invalid ip in address: %q", ipaddr)
315 isv4 := ip.To4() != nil
316 isAllowedSloppyIPv6Submission := func() bool {
317 // Mail user agents that submit are relatively likely to use IPs in EHLO and forget
318 // that an IPv6 address needs to be tagged as such. We can forgive them. For
319 // SMTP servers we are strict.
320 return isehlo && p.conn.submission && !mox.Pedantic && ip.To16() != nil
323 p.xerrorf("ip address is not ipv6")
324 } else if !ipv6 && !isv4 && !isAllowedSloppyIPv6Submission() {
325 if ip.To16() != nil {
326 p.xerrorf("ip address is ipv6, must use syntax [IPv6:...]")
328 p.xerrorf("ip address is not ipv4")
331 return dns.IPDomain{IP: ip}
333 return dns.IPDomain{Domain: p.xdomain()}
336// todo: reduce duplication between implementations: ../smtp/address.go:/xlocalpart ../dkim/parser.go:/xlocalpart ../smtpserver/parse.go:/xlocalpart
337func (p *parser) xlocalpart() smtp.Localpart {
340 if p.hasPrefix(`"`) {
341 s = p.xquotedString(true)
345 s += "." + p.xatom(true)
348 // In the wild, some services use large localparts for generated (bounce) addresses.
349 if mox.Pedantic && len(s) > 64 || len(s) > 128 {
351 p.xerrorf("localpart longer than 64 octets")
353 return smtp.Localpart(norm.NFC.String(s))
357func (p *parser) xquotedString(islocalpart bool) string {
364 if c >= ' ' && c < 0x7f {
369 p.xerrorf("invalid localpart, bad escaped char %c", c)
379 if islocalpart && c > 0x7f && !p.smtputf8 {
380 p.xutf8localparterrorf()
382 if c >= ' ' && c < 0x7f && c != '\\' && c != '"' || (c > 0x7f && p.smtputf8) {
386 p.xerrorf("invalid localpart, invalid character %c", c)
390func (p *parser) xchar() rune {
391 // We are careful to track invalid utf-8 properly.
393 p.xerrorf("need another character")
397 for i, c := range p.orig[p.o:] {
413func (p *parser) xatom(islocalpart bool) string {
414 return p.xtakefn1("atom", func(c rune, i int) bool {
416 case '!', '#', '$', '%', '&', '\'', '*', '+', '-', '/', '=', '?', '^', '_', '`', '{', '|', '}', '~':
419 if islocalpart && c > 0x7f && !p.smtputf8 {
420 p.xutf8localparterrorf()
422 return c >= '0' && c <= '9' || c >= 'A' && c <= 'Z' || (c > 0x7f && p.smtputf8)
427func (p *parser) xstring() string {
428 if p.peekchar() == '"' {
429 return p.xquotedString(false)
431 return p.xatom(false)
435func (p *parser) xparamKeyword() string {
436 return p.xtakefn1("parameter keyword", func(c rune, i int) bool {
437 return c >= '0' && c <= '9' || c >= 'A' && c <= 'Z' || (i > 0 && c == '-')
442func (p *parser) xparamValue() string {
443 return p.xtakefn1("parameter value", func(c rune, i int) bool {
444 return c > ' ' && c < 0x7f && c != '=' || (c > 0x7f && p.smtputf8)
448// for smtp parameters that take a numeric parameter with specified number of
449// digits, eg SIZE=... for MAIL FROM.
450func (p *parser) xnumber(maxDigits int, allowZero bool) int64 {
451 s := p.xtakefn1("number", func(c rune, i int) bool {
452 return (c >= '1' && c <= '9' || c == '0' && (i > 0 || allowZero)) && i < maxDigits
454 v, err := strconv.ParseInt(s, 10, 64)
456 p.xerrorf("bad number %q: %s", s, err)
462func (p *parser) xdatetimeutc() (time.Time, string) {
464 xdash := func() string {
468 xcolon := func() string {
472 xdigits := func(n int) string {
473 s := p.xtakefn1("digits", func(c rune, i int) bool {
474 return c >= '0' && c <= '9' && i < n
477 p.xerrorf("parsing date-time: got %d digits, need %d", len(s), n)
481 s := xdigits(4) + xdash() + xdigits(2) + xdash() + xdigits(2)
482 if !p.hasPrefix("T") {
483 p.xerrorf("expected T for date-time separator")
485 s += p.xtaken(1) + xdigits(2) + xcolon() + xdigits(2) + xcolon() + xdigits(2)
486 layout := time.RFC3339
488 layout = time.RFC3339Nano
489 s += "." + p.xtakefn1("digits", func(c rune, i int) bool {
490 return c >= '0' && c <= '9'
493 if !p.hasPrefix("Z") {
494 p.xerrorf("expected Z for date-time utc timezone")
498 t, err := time.Parse(layout, s)
500 p.xerrorf("bad utc date-time %q: %s", s, err)
505// sasl mechanism, for AUTH command.
507func (p *parser) xsaslMech() string {
508 return p.xtakefn1case("sasl-mech", func(c rune, i int) bool {
509 return i < 20 && (c >= 'A' && c <= 'Z' || c >= '0' && c <= '9' || c == '-' || c == '_')
514func (p *parser) xtext() string {
518 if b >= 0x21 && b < 0x7f && b != '+' && b != '=' && b != ' ' {
528 for _, b := range x {
529 if b >= '0' && b <= '9' || b >= 'A' && b <= 'F' {
532 p.xerrorf("parsing xtext: invalid hexadecimal %q", x)
534 const hex = "0123456789ABCDEF"
535 b = byte(strings.IndexByte(hex, x[0])<<4) | byte(strings.IndexByte(hex, x[1])<<0)