From 1b753b0d53744a6e95467c6bd128ead0edb273cc Mon Sep 17 00:00:00 2001 From: Nicolas Martinelli Date: Wed, 30 Jan 2019 12:50:27 +0000 Subject: [PATCH] [FIX] document: PDF indexing Since PyPDF2 gives bad results for PDF indexing, stop using it since it raises more issues than it helps users. opw-1928446 closes odoo/odoo#30692 --- addons/document/models/ir_attachment.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/addons/document/models/ir_attachment.py b/addons/document/models/ir_attachment.py index 2bc6aee234d..bdc4559e893 100644 --- a/addons/document/models/ir_attachment.py +++ b/addons/document/models/ir_attachment.py @@ -9,7 +9,7 @@ import zipfile from odoo import api, models _logger = logging.getLogger(__name__) -FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc', 'pdf'] +FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc'] def textToString(element): buff = u"" @@ -92,6 +92,9 @@ class IrAttachment(models.Model): def _index_pdf(self, bin_data): '''Index PDF documents''' + # extractText gives very bad results for indexing, hence we don't index PDF anymore. A + # better alternative is probably PDFMiner.six, but not for stable. + # See POC at https://github.com/odoo/odoo/pull/27568. buf = u"" if bin_data.startswith(b'%PDF-'): f = io.BytesIO(bin_data)