[FIX] base: ignore spaces around text content when matching translation

When the same text appears several times inside a translated field but
nested in different HTMLs, the matching for each one is done
independently if various spacing appear in the HTML.

This commit strips the spaces around the matched texts so that texts
that are synchronized on purpose do not become desynchronized.
Doing this leads to collisions on the keys of `text2term`, it therefore
also has to replace it with a dictionary of text to list of terms.

opw-3098819

closes odoo/odoo#109798

X-original-commit: 6cec590a2dad43063d3bb747838393a8df13aa04
Signed-off-by: Benoit Socias (bso) <bso@odoo.com>
This commit is contained in:
Benoit Socias
2023-01-13 17:32:35 +01:00
parent 6721a64cdf
commit 0bc3ae0ed5
3 changed files with 59 additions and 4 deletions
+51
View File
@@ -874,6 +874,57 @@ class TestXMLTranslation(TransactionCase):
self.assertEqual(view.with_env(env_fr).arch_db, archf % (terms_en[0], terms_fr[1]))
self.assertEqual(view.with_env(env_nl).arch_db, archf % (terms_en[0], terms_nl[1]))
def test_sync_xml_collision(self):
""" Check translations of 'arch' after xml tags changes in source terms
when the same term appears in different elements with different
styles.
"""
archf = '''<form class="row">
%s
<div class="s_table_of_content_vertical_navbar" data-name="Navbar" contenteditable="false">
<div class="s_table_of_content_navbar" style="top: 76px;"><a href="#table_of_content_heading_1672668075678_4" class="table_of_content_link">%s</a></div>
</div>
<div class="s_table_of_content_main" data-name="Content">
<section class="pb16">
<h1 data-anchor="true" class="o_default_snippet_text" id="table_of_content_heading_1672668075678_4">%s</h1>
</section>
</div>
</form>'''
terms_en = ('Bread and cheese', 'Knive and Fork', 'Knive <span style="font-weight:bold">and</span> Fork')
terms_fr = ('Pain et fromage', 'Couteau et Fourchette', 'Couteau et Fourchette')
terms_nl = ('Brood and kaas', 'Mes en Vork', 'Mes en Vork')
view = self.create_view(archf, terms_en, en_US=terms_en, fr_FR=terms_fr, nl_NL=terms_nl)
env_nolang = self.env(context={})
env_en = self.env(context={'lang': 'en_US'})
env_fr = self.env(context={'lang': 'fr_FR'})
env_nl = self.env(context={'lang': 'nl_NL'})
self.assertEqual(view.with_env(env_nolang).arch_db, archf % terms_en)
self.assertEqual(view.with_env(env_en).arch_db, archf % terms_en)
self.assertEqual(view.with_env(env_fr).arch_db, archf % terms_fr)
self.assertEqual(view.with_env(env_nl).arch_db, archf % terms_nl)
# modify source term in view (small change)
terms_en = ('Bread and cheese', 'Knife and Fork', 'Knife <span style="font-weight:bold">and</span> Fork')
view.with_env(env_en).write({'arch_db': archf % terms_en})
# check whether translations have been kept
self.assertEqual(view.with_env(env_nolang).arch_db, archf % terms_en)
self.assertEqual(view.with_env(env_en).arch_db, archf % terms_en)
self.assertEqual(view.with_env(env_fr).arch_db, archf % terms_fr)
self.assertEqual(view.with_env(env_nl).arch_db, archf % terms_nl)
# modify source term in view (actual text change)
terms_en = ('Bread and cheese', 'Fork and Knife', 'Fork <span style="font-weight:bold">and</span> Knife')
view.with_env(env_en).write({'arch_db': archf % terms_en})
# check whether translations have been reset
self.assertEqual(view.with_env(env_nolang).arch_db, archf % terms_en)
self.assertEqual(view.with_env(env_en).arch_db, archf % terms_en)
self.assertEqual(view.with_env(env_fr).arch_db, archf % (terms_fr[0], terms_en[1], terms_en[2]))
self.assertEqual(view.with_env(env_nl).arch_db, archf % (terms_nl[0], terms_en[1], terms_en[2]))
def test_cache_consistency(self):
view = self.env["ir.ui.view"].create({
"name": "test_translate_xml_cache_invalidation",
+6 -3
View File
@@ -1801,14 +1801,17 @@ class _String(Field):
continue
from_lang_value = old_translations.get(lang, old_translations.get('en_US'))
translation_dictionary = self.get_translation_dictionary(from_lang_value, old_translations)
text2term = {self.get_text_content(term): term for term in new_terms}
text2terms = defaultdict(list)
for term in new_terms:
text2terms[self.get_text_content(term)].append(term)
for old_term in list(translation_dictionary.keys()):
if old_term not in new_terms:
old_term_text = self.get_text_content(old_term)
matches = get_close_matches(old_term_text, text2term, 1, 0.9)
matches = get_close_matches(old_term_text, text2terms, 1, 0.9)
if matches:
translation_dictionary[text2term[matches[0]]] = translation_dictionary.pop(old_term)
closest_term = get_close_matches(old_term, text2terms[matches[0]], 1, 0)[0]
translation_dictionary[closest_term] = translation_dictionary.pop(old_term)
# pylint: disable=not-callable
new_translations = {
l: self.translate(lambda term: translation_dictionary.get(term, {l: None})[l], cache_value)
+2 -1
View File
@@ -358,7 +358,8 @@ def html_term_converter(value):
def get_text_content(term):
""" Return the textual content of the given term. """
return html.fromstring(term).text_content()
content = html.fromstring(term).text_content()
return " ".join(content.split())
xml_translate.get_text_content = get_text_content
html_translate.get_text_content = get_text_content