When a user sets up a domain name on Odoo, we consider that he has a configuration that makes only one site visible. To do this, the standard solution is to have the following redirections: - http://example.com => https://example.com - https://www.example.com => https://example.com - http://www.example.com => https://example.com It happens that users enter something other than https://example.com in the setting to define the domain names of their websites. If this is the case, [this other commit] would cause an error: - The page indexing bot went to https://example.com but since this was not what was set in the settings, a no index was added so that the page was not referenced. - As soon as the indexing bot went to http://example.com, https://www.example.com or http://www.example.com, it was redirected to https://example.com. As a result, the client ended up with a non-indexed website. The purpose of [this other commit] was just to prevent double indexation of websites (the https://example.odoo.com and the https://example.com). After this commit, the pages will be indexed even if the scheme is not the same as the one specified in the settings. The same goes for the www. which is also ignored. Note that this can have an undesirable effect if the client has a bad configuration and has several sites exposed (https://example.com and https://www.example.com for example). If this is the case, he will end up with a site that is indexed twice. [this other commit]: https://github.com/odoo/odoo/commit/3739d74afe824554b37b1b52ed32ada33692c01a task-3110888 closes odoo/odoo#119819 X-original-commit: c75b35b24c868a821089cafd990c15357cf737e7 Signed-off-by: Romain Derie (rde) <rde@odoo.com>
189 lines
5.9 KiB
Python
189 lines
5.9 KiB
Python
# Part of Odoo. See LICENSE file for full copyright and licensing details.
|
|
import contextlib
|
|
import re
|
|
import werkzeug.urls
|
|
from lxml import etree
|
|
from unittest.mock import Mock, MagicMock, patch
|
|
|
|
from werkzeug.exceptions import NotFound
|
|
from werkzeug.test import EnvironBuilder
|
|
|
|
import odoo
|
|
from odoo.tests.common import HttpCase, HOST
|
|
from odoo.tools.misc import DotDict, frozendict
|
|
|
|
|
|
@contextlib.contextmanager
|
|
def MockRequest(
|
|
env, *, path='/mockrequest', routing=True, multilang=True,
|
|
context=frozendict(), cookies=frozendict(), country_code=None,
|
|
website=None, remote_addr=HOST, environ_base=None,
|
|
# website_sale
|
|
sale_order_id=None, website_sale_current_pl=None,
|
|
):
|
|
|
|
lang_code = context.get('lang', env.context.get('lang', 'en_US'))
|
|
env = env(context=dict(context, lang=lang_code))
|
|
request = Mock(
|
|
# request
|
|
httprequest=Mock(
|
|
host='localhost',
|
|
path=path,
|
|
app=odoo.http.root,
|
|
environ=dict(
|
|
EnvironBuilder(
|
|
path=path,
|
|
base_url=HttpCase.base_url(),
|
|
environ_base=environ_base,
|
|
).get_environ(),
|
|
REMOTE_ADDR=remote_addr,
|
|
),
|
|
cookies=cookies,
|
|
referrer='',
|
|
remote_addr=remote_addr,
|
|
),
|
|
type='http',
|
|
future_response=odoo.http.FutureResponse(),
|
|
params={},
|
|
redirect=env['ir.http']._redirect,
|
|
session=DotDict(
|
|
odoo.http.get_default_session(),
|
|
geoip={'country_code': country_code},
|
|
sale_order_id=sale_order_id,
|
|
website_sale_current_pl=website_sale_current_pl,
|
|
),
|
|
geoip=odoo.http.GeoIP('127.0.0.1'),
|
|
db=env.registry.db_name,
|
|
env=env,
|
|
registry=env.registry,
|
|
cr=env.cr,
|
|
uid=env.uid,
|
|
context=env.context,
|
|
lang=env['res.lang']._lang_get(lang_code),
|
|
website=website,
|
|
)
|
|
if website:
|
|
request.website_routing = website.id
|
|
|
|
# The following code mocks match() to return a fake rule with a fake
|
|
# 'routing' attribute (routing=True) or to raise a NotFound
|
|
# exception (routing=False).
|
|
#
|
|
# router = odoo.http.root.get_db_router()
|
|
# rule, args = router.bind(...).match(path)
|
|
# # arg routing is True => rule.endpoint.routing == {...}
|
|
# # arg routing is False => NotFound exception
|
|
router = MagicMock()
|
|
match = router.return_value.bind.return_value.match
|
|
if routing:
|
|
match.return_value[0].routing = {
|
|
'type': 'http',
|
|
'website': True,
|
|
'multilang': multilang
|
|
}
|
|
else:
|
|
match.side_effect = NotFound
|
|
|
|
def update_context(**overrides):
|
|
request.context = dict(request.context, **overrides)
|
|
|
|
request.update_context = update_context
|
|
|
|
with contextlib.ExitStack() as s:
|
|
odoo.http._request_stack.push(request)
|
|
s.callback(odoo.http._request_stack.pop)
|
|
s.enter_context(patch('odoo.http.root.get_db_router', router))
|
|
|
|
yield request
|
|
|
|
# Fuzzy matching tools
|
|
|
|
def distance(s1="", s2="", limit=4):
|
|
"""
|
|
Limited Levenshtein-ish distance (inspired from Apache text common)
|
|
Note: this does not return quick results for simple cases (empty string, equal strings)
|
|
those checks should be done outside loops that use this function.
|
|
|
|
:param s1: first string
|
|
:param s2: second string
|
|
:param limit: maximum distance to take into account, return -1 if exceeded
|
|
|
|
:return: number of character changes needed to transform s1 into s2 or -1 if this exceeds the limit
|
|
"""
|
|
BIG = 100000 # never reached integer
|
|
if len(s1) > len(s2):
|
|
s1, s2 = s2, s1
|
|
l1 = len(s1)
|
|
l2 = len(s2)
|
|
if l2 - l1 > limit:
|
|
return -1
|
|
boundary = min(l1, limit) + 1
|
|
p = [i if i < boundary else BIG for i in range(0, l1 + 1)]
|
|
d = [BIG for _ in range(0, l1 + 1)]
|
|
for j in range(1, l2 + 1):
|
|
j2 = s2[j - 1]
|
|
d[0] = j
|
|
range_min = max(1, j - limit)
|
|
range_max = min(l1, j + limit)
|
|
if range_min > 1:
|
|
d[range_min - 1] = BIG
|
|
for i in range(range_min, range_max + 1):
|
|
if s1[i - 1] == j2:
|
|
d[i] = p[i - 1]
|
|
else:
|
|
d[i] = 1 + min(d[i - 1], p[i], p[i - 1])
|
|
p, d = d, p
|
|
return p[l1] if p[l1] <= limit else -1
|
|
|
|
def similarity_score(s1, s2):
|
|
"""
|
|
Computes a score that describes how much two strings are matching.
|
|
|
|
:param s1: first string
|
|
:param s2: second string
|
|
|
|
:return: float score, the higher the more similar
|
|
pairs returning non-positive scores should be considered non similar
|
|
"""
|
|
dist = distance(s1, s2)
|
|
if dist == -1:
|
|
return -1
|
|
set1 = set(s1)
|
|
score = len(set1.intersection(s2)) / len(set1)
|
|
score -= dist / len(s1)
|
|
score -= len(set1.symmetric_difference(s2)) / (len(s1) + len(s2))
|
|
return score
|
|
|
|
def text_from_html(html_fragment, collapse_whitespace=False):
|
|
"""
|
|
Returns the plain non-tag text from an html
|
|
|
|
:param html_fragment: document from which text must be extracted
|
|
|
|
:return: text extracted from the html
|
|
"""
|
|
# lxml requires one single root element
|
|
tree = etree.fromstring('<p>%s</p>' % html_fragment, etree.XMLParser(recover=True))
|
|
content = ' '.join(tree.itertext())
|
|
if collapse_whitespace:
|
|
content = re.sub('\\s+', ' ', content).strip()
|
|
return content
|
|
|
|
def get_base_domain(url, strip_www=False):
|
|
"""
|
|
Returns the domain of a given url without the scheme and the www. and the
|
|
final '/' if any.
|
|
|
|
:param url: url from which the domain must be extracted
|
|
:param strip_www: if True, strip the www. from the domain
|
|
|
|
:return: domain of the url
|
|
"""
|
|
if not url:
|
|
return ''
|
|
|
|
url = werkzeug.urls.url_parse(url).netloc
|
|
if strip_www and url.startswith('www.'):
|
|
url = url[4:]
|
|
return url
|