From dff2e242a5e5288cb3c8408a9b635a4c85034e83 Mon Sep 17 00:00:00 2001 From: Nicolas Martinelli Date: Wed, 30 Jan 2019 12:50:27 +0000 Subject: [PATCH] [FIX] document: PDF indexing Backport from 12.0, commit : 1b753b0d53744a6e95467c6bd128ead0edb273cc note: there was also report of PDF content blocking that when indexed blocked an instance worker indefinitely Since PyPDF2 gives bad results for PDF indexing, stop using it since it raises more issues than it helps users. opw-2044679 closes odoo/odoo#35310 closes odoo/odoo#35525 Signed-off-by: Jorge Pinna Puissant (jpp) --- addons/document/models/ir_attachment.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/addons/document/models/ir_attachment.py b/addons/document/models/ir_attachment.py index 2bc6aee234d..bdc4559e893 100644 --- a/addons/document/models/ir_attachment.py +++ b/addons/document/models/ir_attachment.py @@ -9,7 +9,7 @@ import zipfile from odoo import api, models _logger = logging.getLogger(__name__) -FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc', 'pdf'] +FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc'] def textToString(element): buff = u"" @@ -92,6 +92,9 @@ class IrAttachment(models.Model): def _index_pdf(self, bin_data): '''Index PDF documents''' + # extractText gives very bad results for indexing, hence we don't index PDF anymore. A + # better alternative is probably PDFMiner.six, but not for stable. + # See POC at https://github.com/odoo/odoo/pull/27568. buf = u"" if bin_data.startswith(b'%PDF-'): f = io.BytesIO(bin_data)