[IMP] attachment_indexation: improved PDF text extraction
PyPDF performs badly on many types of PDF documents.
We add a text extraction with pdfminer, which is designed for this task.
Because pdf content extraction was so flaky, it was completely
deactivated by 1b753b0d53. We revert that :-)
closes odoo/odoo#38508
Task: 2152494
Signed-off-by: Sébastien Theys (seb) <seb@odoo.com>
This commit is contained in:
@@ -2,14 +2,16 @@
|
||||
# Part of Odoo. See LICENSE file for full copyright and licensing details.
|
||||
import io
|
||||
import logging
|
||||
import PyPDF2
|
||||
import xml.dom.minidom
|
||||
import zipfile
|
||||
from pdfminer.pdfinterp import PDFResourceManager, PDFPageInterpreter
|
||||
from pdfminer.converter import TextConverter
|
||||
from pdfminer.pdfpage import PDFPage
|
||||
|
||||
from odoo import api, models
|
||||
|
||||
_logger = logging.getLogger(__name__)
|
||||
FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc']
|
||||
FTYPES = ['docx', 'pptx', 'xlsx', 'opendoc', 'pdf']
|
||||
|
||||
def textToString(element):
|
||||
buff = u""
|
||||
@@ -91,17 +93,19 @@ class IrAttachment(models.Model):
|
||||
|
||||
def _index_pdf(self, bin_data):
|
||||
'''Index PDF documents'''
|
||||
|
||||
# extractText gives very bad results for indexing, hence we don't index PDF anymore. A
|
||||
# better alternative is probably PDFMiner.six, but not for stable.
|
||||
# See POC at https://github.com/odoo/odoo/pull/27568.
|
||||
buf = u""
|
||||
if bin_data.startswith(b'%PDF-'):
|
||||
f = io.BytesIO(bin_data)
|
||||
try:
|
||||
pdf = PyPDF2.PdfFileReader(f, overwriteWarnings=False)
|
||||
for page in pdf.pages:
|
||||
buf += page.extractText()
|
||||
resource_manager = PDFResourceManager()
|
||||
with io.StringIO() as content, TextConverter(resource_manager, content) as device:
|
||||
logging.getLogger("pdfminer").setLevel(logging.CRITICAL)
|
||||
interpreter = PDFPageInterpreter(resource_manager, device)
|
||||
|
||||
for page in PDFPage.get_pages(f):
|
||||
interpreter.process_page(page)
|
||||
|
||||
buf = content.getvalue()
|
||||
except Exception:
|
||||
pass
|
||||
return buf
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
from . import test_indexation
|
||||
Binary file not shown.
@@ -0,0 +1,16 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
from odoo.tests.common import TransactionCase, tagged
|
||||
import os
|
||||
|
||||
directory = os.path.dirname(__file__)
|
||||
|
||||
|
||||
@tagged('post_install', '-at_install')
|
||||
class TestCaseIndexation(TransactionCase):
|
||||
|
||||
def test_attachment_pdf_indexation(self):
|
||||
with open(os.path.join(directory, 'files', 'test_content.pdf'), 'rb') as file:
|
||||
pdf = file.read()
|
||||
text = self.env['ir.attachment']._index(pdf, 'application/pdf')
|
||||
self.assertEqual(text, 'TestContent!!\x0c', 'the index content should be correct')
|
||||
@@ -21,6 +21,7 @@ mock==2.0.0
|
||||
num2words==0.5.6
|
||||
ofxparse==0.19
|
||||
passlib==1.7.1
|
||||
pdfminer.six==20181108
|
||||
Pillow==5.4.1 ; python_version < '3.7' or sys_platform != 'win32'
|
||||
Pillow==6.1.0 ; sys_platform == 'win32' and python_version >= '3.7'
|
||||
polib==1.1.0
|
||||
|
||||
Reference in New Issue
Block a user