[FIX] website: ignore the scheme for page indexing
When a user sets up a domain name on Odoo, we consider that he has a configuration that makes only one site visible. To do this, the standard solution is to have the following redirections: - http://example.com => https://example.com - https://www.example.com => https://example.com - http://www.example.com => https://example.com It happens that users enter something other than https://example.com in the setting to define the domain names of their websites. If this is the case, [this other commit] would cause an error: - The page indexing bot went to https://example.com but since this was not what was set in the settings, a no index was added so that the page was not referenced. - As soon as the indexing bot went to http://example.com, https://www.example.com or http://www.example.com, it was redirected to https://example.com. As a result, the client ended up with a non-indexed website. The purpose of [this other commit] was just to prevent double indexation of websites (the https://example.odoo.com and the https://example.com). After this commit, the pages will be indexed even if the scheme is not the same as the one specified in the settings. The same goes for the www. which is also ignored. Note that this can have an undesirable effect if the client has a bad configuration and has several sites exposed (https://example.com and https://www.example.com for example). If this is the case, he will end up with a site that is indexed twice. [this other commit]: https://github.com/odoo/odoo/commit/3739d74afe824554b37b1b52ed32ada33692c01a task-3110888 closes odoo/odoo#119819 X-original-commit: c75b35b24c868a821089cafd990c15357cf737e7 Signed-off-by: Romain Derie (rde) <rde@odoo.com>
This commit is contained in:
@@ -28,6 +28,7 @@ from odoo.addons.http_routing.models.ir_http import slug, slugify, _guess_mimety
|
||||
from odoo.addons.portal.controllers.portal import pager as portal_pager
|
||||
from odoo.addons.portal.controllers.web import Home
|
||||
from odoo.addons.web.controllers.binary import Binary
|
||||
from odoo.addons.website.tools import get_base_domain
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -141,7 +142,7 @@ class Website(Home):
|
||||
|
||||
if not isredir and website.domain:
|
||||
domain_from = request.httprequest.environ.get('HTTP_HOST', '')
|
||||
domain_to = werkzeug.urls.url_parse(website.domain).netloc
|
||||
domain_to = get_base_domain(website.domain)
|
||||
if domain_from != domain_to:
|
||||
# redirect to correct domain for a correct routing map
|
||||
url_to = werkzeug.urls.url_join(website.domain, '/website/force/%s?isredir=1&path=%s' % (website.id, path))
|
||||
|
||||
@@ -19,7 +19,7 @@ from markupsafe import Markup
|
||||
from odoo import api, fields, models, tools, http, release, registry
|
||||
from odoo.addons.http_routing.models.ir_http import RequestUID, slugify, url_for
|
||||
from odoo.addons.website.models.ir_http import sitemap_qs2dom
|
||||
from odoo.addons.website.tools import similarity_score, text_from_html
|
||||
from odoo.addons.website.tools import similarity_score, text_from_html, get_base_domain
|
||||
from odoo.addons.portal.controllers.portal import pager
|
||||
from odoo.addons.iap.tools import iap_tools
|
||||
from odoo.exceptions import AccessError, MissingError, UserError, ValidationError
|
||||
@@ -312,6 +312,20 @@ class Website(models.Model):
|
||||
configurator_action_todo = self.env.ref('website.website_configurator_todo')
|
||||
return configurator_action_todo.action_launch()
|
||||
|
||||
def _is_indexable_url(self, url):
|
||||
"""
|
||||
Returns True if the given url has to be indexed by search engines.
|
||||
It is considered that the website must be indexed if the domain name
|
||||
matches the URL. We check if they are equal while ignoring the www. and
|
||||
http(s). This is to index the site even if the user put the www. in the
|
||||
settings while he has a configuration that redirects the www. to the
|
||||
naked domain for example (same thing for http and https).
|
||||
|
||||
:param url: the url to check
|
||||
:return: True if the url has to be indexed, False otherwise
|
||||
"""
|
||||
return get_base_domain(url, True) == get_base_domain(self.domain, True)
|
||||
|
||||
# ----------------------------------------------------------
|
||||
# Configurator
|
||||
# ----------------------------------------------------------
|
||||
@@ -971,7 +985,7 @@ class Website(models.Model):
|
||||
def _filter_domain(website, domain_name, ignore_port=False):
|
||||
"""Ignore `scheme` from the `domain`, just match the `netloc` which
|
||||
is host:port in the version of `url_parse` we use."""
|
||||
website_domain = urls.url_parse(website.domain or '').netloc
|
||||
website_domain = get_base_domain(website.domain)
|
||||
if ignore_port:
|
||||
website_domain = _remove_port(website_domain)
|
||||
domain_name = _remove_port(domain_name)
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# Part of Odoo. See LICENSE file for full copyright and licensing details.
|
||||
import contextlib
|
||||
import re
|
||||
import werkzeug.urls
|
||||
from lxml import etree
|
||||
from unittest.mock import Mock, MagicMock, patch
|
||||
|
||||
@@ -167,3 +168,21 @@ def text_from_html(html_fragment, collapse_whitespace=False):
|
||||
if collapse_whitespace:
|
||||
content = re.sub('\\s+', ' ', content).strip()
|
||||
return content
|
||||
|
||||
def get_base_domain(url, strip_www=False):
|
||||
"""
|
||||
Returns the domain of a given url without the scheme and the www. and the
|
||||
final '/' if any.
|
||||
|
||||
:param url: url from which the domain must be extracted
|
||||
:param strip_www: if True, strip the www. from the domain
|
||||
|
||||
:return: domain of the url
|
||||
"""
|
||||
if not url:
|
||||
return ''
|
||||
|
||||
url = werkzeug.urls.url_parse(url).netloc
|
||||
if strip_www and url.startswith('www.'):
|
||||
url = url[4:]
|
||||
return url
|
||||
|
||||
@@ -91,7 +91,7 @@
|
||||
<t t-set="website_meta" t-value="seo_object and seo_object.get_website_meta() or {}"/>
|
||||
<meta name="default_title" t-att-content="default_title" groups="website.group_website_designer"/>
|
||||
<meta t-if="(main_object and 'website_indexed' in main_object
|
||||
and not main_object.website_indexed) or (website.domain and website.domain not in request.httprequest.url_root)" name="robots" content="noindex"/>
|
||||
and not main_object.website_indexed) or (website.domain and not website._is_indexable_url(request.httprequest.url_root))" name="robots" content="noindex"/>
|
||||
<t t-set="seo_object" t-value="seo_object or main_object"/>
|
||||
<t t-set="meta_description" t-value="seo_object and 'website_meta_description' in seo_object
|
||||
and seo_object.website_meta_description or website_meta_description or website_meta.get('meta_description', '')"/>
|
||||
@@ -2390,7 +2390,7 @@
|
||||
<template id="robots">
|
||||
<t t-translation="off">
|
||||
User-agent: *
|
||||
<t t-if="website.domain and website.domain not in url_root">
|
||||
<t t-if="website.domain and not website._is_indexable_url(url_root)">
|
||||
Disallow: /
|
||||
</t>
|
||||
<t t-else="">
|
||||
|
||||
Reference in New Issue
Block a user