1package message
2
3import (
4 "bufio"
5 "errors"
6 "fmt"
7 "io"
8 "regexp"
9 "slices"
10 "strings"
11
12 "golang.org/x/net/html"
13 "golang.org/x/net/html/atom"
14
15 "github.com/mjl-/mox/mlog"
16 "github.com/mjl-/mox/moxio"
17)
18
19// Preview returns a message preview, based on the first text/plain or text/html
20// part of the message that has textual content. Preview returns at most 256
21// characters (possibly more bytes). Callers may want to truncate and trim trailing
22// whitespace before using the preview.
23//
24// Preview logs at debug level for invalid messages. An error is only returned for
25// serious errors, like i/o errors.
26func (p Part) Preview(log mlog.Log) (string, error) {
27 // ../rfc/8970:190
28
29 // Don't use if Content-Disposition attachment.
30 disp, _, err := p.DispositionFilename()
31 if err != nil {
32 log.Debugx("parsing disposition/filename", err)
33 } else if strings.EqualFold(disp, "attachment") {
34 return "", nil
35 }
36
37 mt := p.MediaType + "/" + p.MediaSubType
38 switch mt {
39 case "TEXT/PLAIN", "/":
40 r := &moxio.LimitReader{R: p.ReaderUTF8OrBinary(), Limit: 1024 * 1024}
41 s, err := previewText(r)
42 if err != nil {
43 if errors.Is(err, moxio.ErrLimit) {
44 log.Debug("no preview in first mb of text message")
45 return "", nil
46 }
47 return "", fmt.Errorf("making preview from text part: %v", err)
48 }
49 return s, nil
50
51 case "TEXT/HTML":
52 r := &moxio.LimitReader{R: p.ReaderUTF8OrBinary(), Limit: 1024 * 1024}
53
54 // First turn the HTML into text.
55 s, err := previewHTML(r)
56 if err != nil {
57 log.Debugx("parsing html part for preview (ignored)", err)
58 return "", nil
59 }
60
61 // Turn text body into a preview text.
62 s, err = previewText(strings.NewReader(s))
63 if err != nil {
64 if errors.Is(err, moxio.ErrLimit) {
65 log.Debug("no preview in first mb of html message")
66 return "", nil
67 }
68 return "", fmt.Errorf("making preview from text from html: %v", err)
69 }
70 return s, nil
71
72 case "MULTIPART/ENCRYPTED":
73 return "", nil
74 }
75
76 for i, sp := range p.Parts {
77 if mt == "MULTIPART/SIGNED" && i >= 1 {
78 break
79 }
80 s, err := sp.Preview(log)
81 if err != nil || s != "" {
82 return s, err
83 }
84 }
85 return "", nil
86}
87
88// previewText returns a line the client can display next to the subject line
89// in a mailbox. It will replace quoted text, and any prefixing "On ... wrote:"
90// line with "[...]" so only new and useful information will be displayed.
91// Trailing signatures are not included.
92func previewText(r io.Reader) (string, error) {
93 // We look quite a bit of lines ahead for trailing signatures with trailing empty lines.
94 var lines []string
95 br := bufio.NewReader(r)
96 ensureLines := func() error {
97 for len(lines) < 10 {
98 line, err := br.ReadString('\n')
99 if line != "" {
100 lines = append(lines, strings.TrimSpace(line))
101 }
102 if err == io.EOF {
103 break
104 } else if err != nil {
105 return fmt.Errorf("read: %w", err)
106 }
107 }
108 return nil
109 }
110 if err := ensureLines(); err != nil {
111 return "", err
112 }
113
114 isSnipped := func(s string) bool {
115 return s == "[...]" || s == "[…]" || s == "..."
116 }
117
118 nextLineQuoted := func(i int) bool {
119 if i+1 < len(lines) && lines[i+1] == "" {
120 i++
121 }
122 return i+1 < len(lines) && (strings.HasPrefix(lines[i+1], ">") || isSnipped(lines[i+1]))
123 }
124
125 // Remainder is signature if we see a line with only and minimum 2 dashes, and
126 // there are no more empty lines, and there aren't more than 5 lines left.
127 isSignature := func() bool {
128 if len(lines) == 0 || !strings.HasPrefix(lines[0], "--") || strings.Trim(strings.TrimSpace(lines[0]), "-") != "" {
129 return false
130 }
131 l := lines[1:]
132 for len(l) > 0 && l[len(l)-1] == "" {
133 l = l[:len(l)-1]
134 }
135 if len(l) >= 5 {
136 return false
137 }
138 return !slices.Contains(l, "")
139 }
140
141 result := ""
142
143 resultSnipped := func() bool {
144 return strings.HasSuffix(result, "[...]\n") || strings.HasSuffix(result, "[…]")
145 }
146
147 // Quick check for initial wrapped "On ... wrote:" line.
148 if len(lines) > 3 && strings.HasPrefix(lines[0], "On ") && !strings.HasSuffix(lines[0], "wrote:") && strings.HasSuffix(lines[1], ":") && nextLineQuoted(1) {
149 result = "[...]\n"
150 lines = lines[3:]
151 if err := ensureLines(); err != nil {
152 return "", err
153 }
154 }
155
156 var err error // set by ensureLines after each loop
157 for ; len(lines) > 0 && !isSignature(); err = ensureLines() {
158 // handle error from ensureLines
159 if err != nil {
160 return "", err
161 }
162
163 line := lines[0]
164 if strings.HasPrefix(line, ">") {
165 if !resultSnipped() {
166 result += "[...]\n"
167 }
168 lines = lines[1:]
169 continue
170 }
171 if line == "" {
172 lines = lines[1:]
173 continue
174 }
175 // Check for a "On <date>, <person> wrote:", we require digits before a quoted
176 // line, with an optional empty line in between. If we don't have any text yet, we
177 // don't require the digits.
178 if strings.HasSuffix(line, ":") && (strings.ContainsAny(line, "0123456789") || result == "") && nextLineQuoted(0) {
179 if !resultSnipped() {
180 result += "[...]\n"
181 }
182 lines = lines[1:]
183 continue
184 }
185 // Skip possibly duplicate snipping by author.
186 if !isSnipped(line) || !resultSnipped() {
187 result += line + "\n"
188 }
189 lines = lines[1:]
190 if len(result) > 250 {
191 break
192 }
193 }
194
195 // Limit number of characters (not bytes). ../rfc/8970:200
196 // To 256 characters. ../rfc/8970:211
197 var o, n int
198 for o = range result {
199 n++
200 if n > 256 {
201 result = result[:o]
202 break
203 }
204 }
205
206 return result, nil
207}
208
209// Any text inside these html elements (recursively) is ignored.
210var ignoreAtoms = atomMap(
211 atom.Dialog,
212 atom.Head,
213 atom.Map,
214 atom.Math,
215 atom.Script,
216 atom.Style,
217 atom.Svg,
218 atom.Template,
219)
220
221// Inline elements don't force newlines at beginning & end of text in this element.
222// https://developer.mozilla.org/en-US/docs/Web/HTML/Element#inline_text_semantics
223var inlineAtoms = atomMap(
224 atom.A,
225 atom.Abbr,
226 atom.B,
227 atom.Bdi,
228 atom.Bdo,
229 atom.Cite,
230 atom.Code,
231 atom.Data,
232 atom.Dfn,
233 atom.Em,
234 atom.I,
235 atom.Kbd,
236 atom.Mark,
237 atom.Q,
238 atom.Rp,
239 atom.Rt,
240 atom.Ruby,
241 atom.S,
242 atom.Samp,
243 atom.Small,
244 atom.Span,
245 atom.Strong,
246 atom.Sub,
247 atom.Sup,
248 atom.Time,
249 atom.U,
250 atom.Var,
251 atom.Wbr,
252
253 atom.Del,
254 atom.Ins,
255
256 // We treat these specially, inserting a space after them instead of a newline.
257 atom.Td,
258 atom.Th,
259)
260
261func atomMap(l ...atom.Atom) map[atom.Atom]bool {
262 m := map[atom.Atom]bool{}
263 for _, a := range l {
264 m[a] = true
265 }
266 return m
267}
268
269var regexpSpace = regexp.MustCompile(`[ \t]+`) // Replaced with single space.
270var regexpNewline = regexp.MustCompile(`\n\n\n+`) // Replaced with single newline.
271var regexpZeroWidth = regexp.MustCompile("[\u00a0\u200b\u200c\u200d][\u00a0\u200b\u200c\u200d]+") // Removed, combinations don't make sense, generated.
272
273func previewHTML(r io.Reader) (string, error) {
274 // Stack/state, based on elements.
275 var ignores []bool
276 var inlines []bool
277
278 var text string // Collecting text.
279 var err error // Set when walking DOM.
280 var quoteLevel int
281
282 // We'll walk the DOM nodes, keeping track of whether we are ignoring text, and
283 // whether we are in an inline or block element, and building up the text. We stop
284 // when we have enough data, returning false in that case.
285 var walk func(n *html.Node) bool
286 walk = func(n *html.Node) bool {
287 switch n.Type {
288 case html.ErrorNode:
289 err = fmt.Errorf("unexpected error node")
290 return false
291
292 case html.ElementNode:
293 ignores = append(ignores, ignoreAtoms[n.DataAtom])
294 inline := inlineAtoms[n.DataAtom]
295 inlines = append(inlines, inline)
296 if n.DataAtom == atom.Blockquote {
297 quoteLevel++
298 }
299 defer func() {
300 if n.DataAtom == atom.Blockquote {
301 quoteLevel--
302 }
303 if !inline && !strings.HasSuffix(text, "\n\n") {
304 text += "\n"
305 } else if (n.DataAtom == atom.Td || n.DataAtom == atom.Th) && !strings.HasSuffix(text, " ") {
306 text += " "
307 }
308
309 ignores = ignores[:len(ignores)-1]
310 inlines = inlines[:len(inlines)-1]
311 }()
312
313 case html.TextNode:
314 if slices.Contains(ignores, true) {
315 return true
316 }
317 // Collapse all kinds of weird whitespace-like characters into a space, except for newline and ignoring carriage return.
318 var s string
319 for _, c := range n.Data {
320 if c == '\r' {
321 continue
322 } else if c == '\t' {
323 s += " "
324 } else {
325 s += string(c)
326 }
327 }
328 s = regexpSpace.ReplaceAllString(s, " ")
329 s = regexpNewline.ReplaceAllString(s, "\n")
330 s = regexpZeroWidth.ReplaceAllString(s, "")
331
332 inline := len(inlines) > 0 && inlines[len(inlines)-1]
333 ts := strings.TrimSpace(s)
334 if !inline && ts == "" {
335 break
336 }
337 if ts != "" || !strings.HasSuffix(s, " ") && !strings.HasSuffix(s, "\n") {
338 if quoteLevel > 0 {
339 q := strings.Repeat("> ", quoteLevel)
340 var sb strings.Builder
341 for s != "" {
342 o := strings.IndexByte(s, '\n')
343 if o < 0 {
344 o = len(s)
345 } else {
346 o++
347 }
348 sb.WriteString(q)
349 sb.WriteString(s[:o])
350 s = s[o:]
351 }
352 s = sb.String()
353 }
354 text += s
355 }
356 // We need to generate at most 256 characters of preview. The text we're gathering
357 // will be cleaned up, with quoting removed, so we'll end up with less. Hopefully,
358 // 4k bytes is enough to read.
359 if len(text) >= 4*1024 {
360 return false
361 }
362 }
363 // Ignored: DocumentNode, CommentNode, DoctypeNode, RawNode
364
365 for cn := range n.ChildNodes() {
366 if !walk(cn) {
367 break
368 }
369 }
370
371 return true
372 }
373
374 node, err := html.Parse(r)
375 if err != nil {
376 return "", fmt.Errorf("parsing html: %v", err)
377 }
378
379 // Build text.
380 walk(node)
381
382 text = strings.TrimSpace(text)
383 text = regexpSpace.ReplaceAllString(text, " ")
384 return text, err
385}
386