[IMP] attachment_indexation: improved PDF text extraction

PyPDF performs badly on many types of PDF documents.
We add a text extraction with pdfminer, which is designed for this task.
Because pdf content extraction was so flaky, it was completely
deactivated by 1b753b0d53. We revert that :-)

closes odoo/odoo#38508

Task: 2152494
Signed-off-by: Sébastien Theys (seb) <seb@odoo.com>
This commit is contained in:
len-odoo
2020-01-29 10:26:46 +00:00
committed by Thanh Dodeur
parent b3661017d0
commit 183af97168
5 changed files with 33 additions and 9 deletions
@@ -2,14 +2,16 @@
# Part of Odoo. See LICENSE file for full copyright and licensing details.
import io
import logging
import PyPDF2
import xml.dom.minidom
import zipfile
from pdfminer.pdfinterp import PDFResourceManager, PDFPageInterpreter
from pdfminer.converter import TextConverter
from pdfminer.pdfpage import PDFPage
from odoo import api, models
_logger = logging.getLogger(__name__)
FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc']
FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc', 'pdf']
def textToString(element):
buff = u""
@@ -91,17 +93,19 @@ class IrAttachment(models.Model):
def _index_pdf(self, bin_data):
'''Index PDF documents'''
# extractText gives very bad results for indexing, hence we don't index PDF anymore. A
# better alternative is probably PDFMiner.six, but not for stable.
# See POC at https://github.com/odoo/odoo/pull/27568.
buf = u""
if bin_data.startswith(b'%PDF-'):
f = io.BytesIO(bin_data)
try:
pdf = PyPDF2.PdfFileReader(f, overwriteWarnings=False)
for page in pdf.pages:
buf += page.extractText()
resource_manager = PDFResourceManager()
with io.StringIO() as content, TextConverter(resource_manager, content) as device:
logging.getLogger("pdfminer").setLevel(logging.CRITICAL)
interpreter = PDFPageInterpreter(resource_manager, device)
for page in PDFPage.get_pages(f):
interpreter.process_page(page)
buf = content.getvalue()
except Exception:
pass
return buf
@@ -0,0 +1,3 @@
# -*- coding: utf-8 -*-
from . import test_indexation
@@ -0,0 +1,16 @@
# -*- coding: utf-8 -*-
from odoo.tests.common import TransactionCase, tagged
import os
directory = os.path.dirname(__file__)
@tagged('post_install', '-at_install')
class TestCaseIndexation(TransactionCase):
def test_attachment_pdf_indexation(self):
with open(os.path.join(directory, 'files', 'test_content.pdf'), 'rb') as file:
pdf = file.read()
text = self.env['ir.attachment']._index(pdf, 'application/pdf')
self.assertEqual(text, 'TestContent!!\x0c', 'the index content should be correct')
+1
View File
@@ -21,6 +21,7 @@ mock==2.0.0
num2words==0.5.6
ofxparse==0.19
passlib==1.7.1
pdfminer.six==20181108
Pillow==5.4.1 ; python_version < '3.7' or sys_platform != 'win32'
Pillow==6.1.0 ; sys_platform == 'win32' and python_version >= '3.7'
polib==1.1.0