[FIX] mail: sanitize incoming mail to remove unwanted class + add test

some incoming mail containing class like "fa-spin" or "modal in" have an unwanted behavior in Odoo, incoming mail should not have
special classes, so we remove all the class that are not in the whitelist.
This commit is contained in:
Cedric Snauwaert
2015-12-09 13:34:30 +01:00
parent afa28f075a
commit c867927b2e
3 changed files with 60 additions and 4 deletions
+7
View File
@@ -330,6 +330,13 @@ class TestCleaner(unittest.TestCase):
for ext in test_mail_examples.BUG_3_OUT:
self.assertNotIn(ext, new_html, 'html_email_cleaner did not removed invalid content')
def test_80_remove_classes(self):
new_html = html_email_clean(test_mail_examples.REMOVE_CLASS, remove=True)
for ext in test_mail_examples.REMOVE_CLASS_IN:
self.assertIn(ext, new_html, 'html_email_cleaner wrongly removed classes')
for ext in test_mail_examples.REMOVE_CLASS_OUT:
self.assertNotIn(ext, new_html, 'html_email_cleaner did not removed correctly unwanted classes')
def test_90_misc(self):
# False boolean for text must return empty string
new_html = html_email_clean(False)
@@ -91,12 +91,12 @@ OERP_WEBSITE_HTML_1 = """
OERP_WEBSITE_HTML_1_IN = [
'Manage your company most important asset: People',
'img class="img-rounded img-responsive" src="/website/static/src/img/china_thumb.jpg"',
'src="/website/static/src/img/china_thumb.jpg"',
]
OERP_WEBSITE_HTML_1_OUT = [
'Break down information silos.',
'Keep track of the vacation days accrued by each employee',
'img class="img-rounded img-responsive" src="/website/static/src/img/deers_thumb.jpg',
'src="/website/static/src/img/deers_thumb.jpg',
]
OERP_WEBSITE_HTML_2 = """
@@ -205,7 +205,7 @@ OERP_WEBSITE_HTML_2_IN = [
]
OERP_WEBSITE_HTML_2_OUT = [
'Make every employee feel more connected',
'img class="img-responsive shadow" src="/website/static/src/img/text_image.png',
'src="/website/static/src/img/text_image.png',
]
TEXT_1 = """I contact you about our meeting tomorrow. Here is the schedule I propose:
@@ -1173,3 +1173,34 @@ BUG_3_IN = [
BUG_3_OUT = [
'New kanban view of documents'
]
REMOVE_CLASS = """
<div style="FONT-SIZE: 12pt; FONT-FAMILY: 'Times New Roman'; COLOR: #000000">
<div>Hello</div>
<div>I have just installed Odoo 9 and I've got the following error:</div>
<div>&nbsp;</div>
<div class="openerp openerp_webclient_container oe_webclient">
<div class="oe_loading" style="DISPLAY: none">&nbsp;</div>
</div>
<div class="modal-backdrop in"></div>
<div role="dialog" tabindex="-1" aria-hidden="false" class="modal in" style="DISPLAY: block" data-backdrop="static">
<div class="modal-dialog modal-lg">
<div class="modal-content openerp">
<div class="modal-header">
<h4 class="modal-title">Odoo Error<span class="o_subtitle text-muted"></span></h4>
</div>
<div class="o_error_detail modal-body">
<pre>An error occured in a modal and I will send you back the html to try opening one on your end</pre>
</div>
</div>
</div>
</div>
</div>
"""
REMOVE_CLASS_IN = [
'An error occured in a modal and I will send you back the html to try opening one on your end'
]
REMOVE_CLASS_OUT = [
'<div class="modal-backdrop in">'
]
+19 -1
View File
@@ -26,6 +26,7 @@ _logger = logging.getLogger(__name__)
tags_to_kill = ["script", "head", "meta", "title", "link", "style", "frame", "iframe", "base", "object", "embed"]
tags_to_remove = ['html', 'body']
whitelist_classes = set(['WordSection1', 'MsoNormal', 'SkyDrivePlaceholder', 'oe_mail_expand', 'stopSpelling'])
# allow new semantic HTML5 tags
allowed_tags = clean.defs.tags | frozenset('article section header footer hgroup nav aside figure main'.split() + [etree.Comment])
@@ -292,6 +293,11 @@ def html_email_clean(html, remove=False, shorten=False, max_length=300, expand_o
if expand_options is None:
expand_options = {}
whitelist_classes_local = whitelist_classes.copy()
if expand_options.get('oe_expand_container_class'):
whitelist_classes_local.add(expand_options.get('oe_expand_container_class'))
if expand_options.get('oe_expand_a_class'):
whitelist_classes_local.add(expand_options.get('oe_expand_a_class'))
if not html or not isinstance(html, basestring):
return html
@@ -336,6 +342,7 @@ def html_email_clean(html, remove=False, shorten=False, max_length=300, expand_o
quoted = False
quote_begin = False
overlength = False
replace_class = False
overlength_section_id = None
overlength_section_count = 0
cur_char_nbr = 0
@@ -347,6 +354,17 @@ def html_email_clean(html, remove=False, shorten=False, max_length=300, expand_o
# do not take into account multiple spaces that are displayed as max 1 space in html
node_text = ' '.join((node.text and node.text.strip(' \t\r\n') or '').split())
# remove unwanted classes from node
if node.get('class'):
sanitize_classes = []
for _class in node.get('class').split(' '):
if _class in whitelist_classes_local:
sanitize_classes.append(_class)
else:
sanitize_classes.append('cleaned_'+_class)
replace_class = True
node.set('class', ' '.join(sanitize_classes))
# root: try to tag the client used to write the html
if 'WordSection1' in node.get('class', '') or 'MsoNormal' in node.get('class', ''):
root.set('msoffice', '1')
@@ -448,7 +466,7 @@ def html_email_clean(html, remove=False, shorten=False, max_length=300, expand_o
node_class = node.get('class', '') + ' oe_mail_cleaned'
node.set('class', node_class)
if not overlength and not quote_begin and not quoted:
if not overlength and not quote_begin and not quoted and not replace_class:
return html
# html: \n that were tail of elements have been encapsulated into <span> -> back to \n