[FIX] mail_thread: save attachment from mail in same encoding
Steps to reproduce:
- Configure incoming mail server and set it to create X record
on incoming mails (X can be any model with a chatter)
- Create a CSV file and set the encoding to UTF-16
- Send the CSV file through Gmail to the Odoo instance
- Go to model X and open the created record
- In the chatter, click/download the CSV file
- Open the downloaded file with Geany (or any file editor that can
show the file encoding)
Issue:
The file encoding is not the same as the original file (utf-8 instead
of utf-16).
Working with Outlook.
Cause:
The difference between Outlook and Gmail is that Gmail provides the
charset of the file.
The content of the mail is retrieved using `email` python lib.
The lib will try to retrieve the charset of the file and fallback
on `ASCII` if not available, then return the decode content.
```python
def get_text_content(msg, errors='replace'):
content = msg.get_payload(decode=True)
charset = msg.get_param('charset', 'ASCII')
return content.decode(charset, errors=errors)
```
Example:
content = b'd\x00a\x00,\x00,\x00,\......'
Outlook:
charset = 'ASCII'
return => 'd\x00a\x00,\x00,\x00...'
Gmail:
charset = 'UTF-16LE'
return => 'da,,,,,\n,,,,,\....'
In the post process of the attachment, the content is encoded in
'utf-8' (to then encoded in base64) before creating the attachment
record.
Content encoded to 'utf-8':
Outlook: b'd\x00a\x00,\x00,\x00...'
Gmail: b'da,,,,,\n,,,,,\n....'
Therefore, when writing the file on the disk, the encoding is based
on the binary content.
Solution:
When parsing the mail, add the encoding charset to the `info` variable.
Then, when creating the attachment, use the charset in `info` (or
fallback on 'utf'8' if no charset set) to encode the content.
opw-3089009
closes odoo/odoo#118694
X-original-commit: 2d1b13e68d8bb204ecfe12a9d3db519c4264592c
Signed-off-by: Nicolas Lempereur (nle) <nle@odoo.com>
Signed-off-by: Nasreddin Boulif (bon) <bon@odoo.com>
This commit is contained in:
@@ -1399,25 +1399,27 @@ class MailThread(models.AbstractModel):
|
||||
filename = part.get_filename() # I may not properly handle all charsets
|
||||
encoding = part.get_content_charset() # None if attachment
|
||||
|
||||
content = part.get_content()
|
||||
info = {'encoding': encoding}
|
||||
# 0) Inline Attachments -> attachments, with a third part in the tuple to match cid / attachment
|
||||
if filename and part.get('content-id'):
|
||||
inner_cid = part.get('content-id').strip('><')
|
||||
attachments.append(self._Attachment(filename, part.get_content(), {'cid': inner_cid}))
|
||||
info['cid'] = part.get('content-id').strip('><')
|
||||
attachments.append(self._Attachment(filename, content, info))
|
||||
continue
|
||||
# 1) Explicit Attachments -> attachments
|
||||
if filename or part.get('content-disposition', '').strip().startswith('attachment'):
|
||||
attachments.append(self._Attachment(filename or 'attachment', part.get_content(), {}))
|
||||
attachments.append(self._Attachment(filename or 'attachment', content, info))
|
||||
continue
|
||||
# 2) text/plain -> <pre/>
|
||||
if part.get_content_type() == 'text/plain' and (not alternative or not body):
|
||||
body = tools.append_content_to_html(body, tools.ustr(part.get_content(),
|
||||
body = tools.append_content_to_html(body, tools.ustr(content,
|
||||
encoding, errors='replace'), preserve=True)
|
||||
# 3) text/html -> raw
|
||||
elif part.get_content_type() == 'text/html':
|
||||
# mutlipart/alternative have one text and a html part, keep only the second
|
||||
# mixed allows several html parts, append html content
|
||||
append_content = not alternative or (html and mixed)
|
||||
html = tools.ustr(part.get_content(), encoding, errors='replace')
|
||||
html = tools.ustr(content, encoding, errors='replace')
|
||||
if not append_content:
|
||||
body = html
|
||||
else:
|
||||
@@ -1426,7 +1428,7 @@ class MailThread(models.AbstractModel):
|
||||
body = tools.html_sanitize(body, sanitize_tags=False, strip_classes=True)
|
||||
# 4) Anything else -> attachment
|
||||
else:
|
||||
attachments.append(self._Attachment(filename or 'attachment', part.get_content(), {}))
|
||||
attachments.append(self._Attachment(filename or 'attachment', content, info))
|
||||
|
||||
return self._message_parse_extract_payload_postprocess(message, {'body': body, 'attachments': attachments})
|
||||
|
||||
@@ -2147,6 +2149,7 @@ class MailThread(models.AbstractModel):
|
||||
if len(attachment) == 2:
|
||||
name, content = attachment
|
||||
cid = False
|
||||
info = {}
|
||||
elif len(attachment) == 3:
|
||||
name, content, info = attachment
|
||||
cid = info and info.get('cid')
|
||||
@@ -2154,7 +2157,8 @@ class MailThread(models.AbstractModel):
|
||||
continue
|
||||
|
||||
if isinstance(content, str):
|
||||
content = content.encode('utf-8')
|
||||
encoding = info and info.get('encoding')
|
||||
content = content.encode(encoding or 'utf-8')
|
||||
elif isinstance(content, EmailMessage):
|
||||
content = content.as_bytes()
|
||||
elif content is None:
|
||||
|
||||
@@ -214,6 +214,39 @@ Content-Type: text/html;
|
||||
--Apple-Mail=_9331E12B-8BD2-4EC7-B53E-01F3FBEC9227--
|
||||
"""
|
||||
|
||||
MAIL_FILE_ENCODING = """MIME-Version: 1.0
|
||||
Date: Sun, 26 Mar 2023 05:23:22 +0200
|
||||
Message-ID: {msg_id}
|
||||
Subject: {subject}
|
||||
From: "Sylvie Lelitre" <test.sylvie.lelitre@agrolait.com>
|
||||
To: groups@test.com
|
||||
Content-Type: multipart/mixed; boundary="000000000000b951de05f7c47a9e"
|
||||
|
||||
--000000000000b951de05f7c47a9e
|
||||
Content-Type: multipart/alternative; boundary="000000000000b951da05f7c47a9c"
|
||||
|
||||
--000000000000b951da05f7c47a9c
|
||||
Content-Type: text/plain; charset="UTF-8"
|
||||
|
||||
Test Body
|
||||
|
||||
--000000000000b951da05f7c47a9c
|
||||
Content-Type: text/html; charset="UTF-8"
|
||||
|
||||
<div dir="ltr">Test Body</div>
|
||||
|
||||
--000000000000b951da05f7c47a9c--
|
||||
--000000000000b951de05f7c47a9e
|
||||
Content-Type: text/plain; name="test.txt"{charset}
|
||||
Content-Disposition: attachment; filename="test.txt"
|
||||
Content-Transfer-Encoding: base64
|
||||
X-Attachment-Id: f_lfosfm0l0
|
||||
Content-ID: <f_lfosfm0l0>
|
||||
|
||||
{content}
|
||||
|
||||
--000000000000b951de05f7c47a9e--
|
||||
"""
|
||||
|
||||
MAIL_MULTIPART_BINARY_OCTET_STREAM = """X-Original-To: raoul@grosbedon.fr
|
||||
Delivered-To: raoul@grosbedon.fr
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
# Part of Odoo. See LICENSE file for full copyright and licensing details.
|
||||
|
||||
import base64
|
||||
import socket
|
||||
|
||||
from datetime import datetime
|
||||
@@ -1585,6 +1586,23 @@ class TestMailgateway(MailCommon):
|
||||
self.assertEqual(record.name, 'Spammy')
|
||||
self.assertEqual(record._name, 'mail.test.gateway')
|
||||
|
||||
@mute_logger('odoo.addons.mail.models.mail_thread')
|
||||
def test_message_process_file_encoding(self):
|
||||
""" Incoming email with file encoding """
|
||||
file_content = 'Hello World'
|
||||
for encoding in ['', 'UTF-8', 'UTF-16LE', 'UTF-32BE']:
|
||||
file_content_b64 = base64.b64encode(file_content.encode(encoding or 'utf-8')).decode()
|
||||
record = self.format_and_process(test_mail_data.MAIL_FILE_ENCODING,
|
||||
self.email_from, 'groups@test.com',
|
||||
subject='Test Charset %s' % encoding or 'Unset',
|
||||
charset='; charset="%s"' % encoding if encoding else '',
|
||||
content=file_content_b64
|
||||
)
|
||||
attachment = record.message_ids.attachment_ids
|
||||
self.assertEqual(file_content, attachment.raw.decode(encoding or 'utf-8'))
|
||||
if encoding not in ['', 'UTF-8']:
|
||||
self.assertNotEqual(file_content, attachment.raw.decode('utf-8'))
|
||||
|
||||
# --------------------------------------------------
|
||||
# Emails loop detection
|
||||
# --------------------------------------------------
|
||||
|
||||
Reference in New Issue
Block a user