excel识别

This commit is contained in:
李鹏宇
2026-06-12 18:10:34 +08:00
parent 09ef1e6277
commit 3194db046b
8 changed files with 256 additions and 72 deletions
+1
View File
@@ -26,6 +26,7 @@
'views/ai_user_profile_views.xml',
'views/ai_token_usage_views.xml',
'views/ai_conversation_views.xml',
'views/ai_import_record_views.xml',
'data/demo_provider.xml',
'data/demo_rule.xml',
'data/demo_manual.xml',
+85 -29
View File
@@ -187,10 +187,11 @@ class AiChatController(http.Controller):
@http.route('/ai/attachment/upload', type='http', auth='user', methods=['POST'], csrf=False)
def attachment_upload(self, **kwargs):
"""上传聊天附件(图片/PDF),调 OCR 提取文字并返回结果。
"""上传聊天附件(图片/PDF/Excel),调 OCR 或 Excel 解析器提取文字并返回结果。
请求:multipart/form-data,字段名 'file'
返回:JSON { success, attachment_id, filename, mimetype, ocr_text, error }
可能携带 conversation_id 用于关联对话
返回:JSON { success, attachment_id, filename, mimetype, ocr_text, error, meta }
"""
err = self._check_ai_access()
if err:
@@ -205,30 +206,50 @@ class AiChatController(http.Controller):
filename = uploaded.filename or 'untitled'
mimetype = uploaded.content_type or 'application/octet-stream'
# 限制文件类型
allowed = {'image/png', 'image/jpeg', 'image/jpg', 'image/bmp',
'image/gif', 'application/pdf'}
if mimetype not in allowed:
# 文件类型分类
IMG_MIMES = {'image/png', 'image/jpeg', 'image/jpg', 'image/bmp', 'image/gif'}
PDF_MIMES = {'application/pdf'}
EXCEL_MIMES = {
'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', # .xlsx
'application/vnd.ms-excel', # .xls
'text/csv', # .csv
'application/vnd.ms-excel.sheet.macroEnabled.12', # .xlsm
}
is_image = mimetype in IMG_MIMES
is_pdf = mimetype in PDF_MIMES
is_excel = mimetype in EXCEL_MIMES
# 对于 Excel,也接受通过文件扩展名判断(企微/浏览器可能给错 MIME)
if not is_excel and not is_image and not is_pdf:
lower_name = filename.lower()
if lower_name.endswith(('.xlsx', '.xls', '.csv', '.xlsm')):
is_excel = True
mimetype = 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet'
if not (is_image or is_pdf or is_excel):
return request.make_json_response({
'success': False,
'error': f'不支持的文件格式:{mimetype},仅支持 PNG/JPG/BMP/GIF/PDF'
'error': f'不支持的文件格式({mimetype}),支持:PNG/JPG/BMP/GIF/PDF/Excel/CSV'
})
# 限制文件大小 (10MB)
max_size = 10 * 1024 * 1024
# 限制文件大小
max_size = 20 * 1024 * 1024 if is_excel else 10 * 1024 * 1024
if len(raw_bytes) > max_size:
size_mb = len(raw_bytes) / 1024 / 1024
return request.make_json_response({
'success': False,
'error': f'文件过大({len(raw_bytes) / 1024 / 1024:.1f}MB),上限 10MB'
'error': f'文件过大({size_mb:.1f}MB),上限 {max_size // 1024 // 1024}MB'
})
# 获取当前对话的 Provider OCR 配置
# 获取当前对话的 Provider 配置
conv_id = kwargs.get('conversation_id')
provider = None
conversation = None
if conv_id:
conv = request.env['ai.conversation'].sudo().browse(int(conv_id))
if conv.exists():
provider = conv.provider_id
conversation = request.env['ai.conversation'].sudo().browse(int(conv_id))
if conversation.exists():
provider = conversation.provider_id
# 查找或创建 AI 聊天附件专用文件夹
Folder = request.env['documents.folder'].sudo()
@@ -236,7 +257,7 @@ class AiChatController(http.Controller):
if not folder:
folder = Folder.create({'name': 'AI 聊天附件'})
# 创建 documents.document(raw 字段接受原始二进制,内部通过 attachment_id 存储)
# 创建 documents.document
document = request.env['documents.document'].sudo().create({
'name': filename,
'mimetype': mimetype,
@@ -244,10 +265,54 @@ class AiChatController(http.Controller):
'folder_id': folder.id,
})
# OCR 提取文字
ocr_text = ''
ocr_error = ''
ocr_diag = {} # 诊断信息
ocr_diag = {}
file_type_code = 'image'
# —— Excel 解析 ——
if is_excel:
file_type_code = 'excel'
ocr_diag['parser'] = 'excel_parser'
# 创建导入记录
import_record = request.env['ai.import.record'].sudo().create({
'document_id': document.id,
'filename': filename,
'file_type': 'excel',
'user_id': request.env.uid,
'ocr_status': 'pending',
})
if conversation:
import_record.write({'conversation_id': conversation.id})
from ..utils.excel_parser import parse_excel_to_markdown, MAX_ROWS
ocr_diag['max_rows'] = MAX_ROWS
markdown_text, meta = parse_excel_to_markdown(raw_bytes, filename)
ocr_diag['meta'] = meta
if meta.get('error'):
ocr_error = meta['error']
import_record.write({'ocr_status': 'failed', 'ocr_error': ocr_error})
else:
ocr_text = markdown_text
document.sudo().write({'ai_ocr_content': ocr_text})
ocr_status = 'success' if not meta.get('truncated') else 'success'
import_record.write({
'ocr_status': ocr_status,
'ocr_content': ocr_text,
'ocr_error': ''
})
# 查找匹配的导入规则
rule = request.env['ai.import.rule'].sudo().find_matching_rule(
filename, 'excel', mimetype
)
if rule:
import_record.write({'import_rule_id': rule.id})
ocr_diag['matched_rule'] = rule.name
_logger.info('[Excel-Upload] 匹配导入规则: %s', rule.name)
# —— 图片/PDF OCR ——
elif is_image or is_pdf:
file_type_code = 'image' if is_image else 'pdf'
if provider and provider.ocr_enabled:
from ..models.ocr_provider import get_ocr_instance
ocr, init_err = get_ocr_instance(provider)
@@ -255,12 +320,9 @@ class AiChatController(http.Controller):
ocr_diag['ocr_enabled'] = True
if ocr is None:
ocr_error = init_err or 'OCR 实例创建失败'
_logger.error('[AI-OCR] get_ocr_instance 失败: %s', ocr_error)
ocr_diag['init_error'] = ocr_error
else:
ocr_diag['init_ok'] = True
# 直接用通用文字识别(不先走票据识别,避免不必要的 API 调用)
_logger.info('[AI-OCR] 开始识别 %s (%d bytes, %s)', filename, len(raw_bytes), mimetype)
result = ocr.recognize_general(raw_bytes, mimetype)
ocr_diag['method'] = 'GeneralBasicOCR'
ocr_diag['success'] = result.get('success')
@@ -269,20 +331,13 @@ class AiChatController(http.Controller):
ocr_diag['text_len'] = len(ocr_text)
if ocr_text:
document.sudo().write({'ai_ocr_content': ocr_text})
_logger.info('[AI-OCR] ✅ 识别成功,%d 字符: %s', len(ocr_text), ocr_text[:200])
else:
ocr_error = '图片中未识别到文字'
_logger.info('[AI-OCR] ⚠️ 识别完成但无文字 — 图片中可能确实没有印刷文字')
else:
ocr_error = result.get('error', 'OCR 调用失败')
ocr_diag['error'] = ocr_error
_logger.error('[AI-OCR] ❌ %s', ocr_error)
else:
ocr_diag['ocr_enabled'] = False
if not provider:
ocr_diag['reason'] = '未找到 AI 服务商(conversation_id 可能无效)'
else:
ocr_diag['reason'] = 'OCR 未启用'
ocr_diag['reason'] = 'OCR 未启用' if provider else '未找到 AI 服务商'
return request.make_json_response({
'success': True,
@@ -291,5 +346,6 @@ class AiChatController(http.Controller):
'mimetype': mimetype,
'ocr_text': ocr_text,
'ocr_error': ocr_error,
'_ocr_diag': ocr_diag, # 前端 console 可查看
'file_type': file_type_code,
'_ocr_diag': ocr_diag,
})
+1
View File
@@ -9,3 +9,4 @@ from . import ai_report
from . import ai_report_template
from . import ai_report_template_config
from . import ocr_provider
from . import ai_import_record
@@ -31,3 +31,7 @@ access_ai_report_template_domain_line_user,查询条件行-用户只读,model_ai
access_ai_report_template_domain_line_admin,查询条件行-管理员,model_ai_report_template_domain_line,yuthon_ai_agent.group_ai_agent_admin,1,1,1,1
access_ai_report_template_chart_user,图表配置-用户只读,model_ai_report_template_chart,yuthon_ai_agent.group_ai_agent_user,1,0,0,0
access_ai_report_template_chart_admin,图表配置-管理员,model_ai_report_template_chart,yuthon_ai_agent.group_ai_agent_admin,1,1,1,1
access_ai_import_record_user,附件导入记录-用户,model_ai_import_record,yuthon_ai_agent.group_ai_agent_user,1,0,1,0
access_ai_import_record_admin,附件导入记录-管理员,model_ai_import_record,yuthon_ai_agent.group_ai_agent_admin,1,1,1,1
access_ai_import_rule_admin,导入规则-管理员,model_ai_import_rule,yuthon_ai_agent.group_ai_agent_admin,1,1,1,1
access_ai_import_rule_user,导入规则-用户只读,model_ai_import_rule,yuthon_ai_agent.group_ai_agent_user,1,0,0,0
1 id name model_id:id group_id:id perm_read perm_write perm_create perm_unlink
31 access_ai_report_template_domain_line_admin 查询条件行-管理员 model_ai_report_template_domain_line yuthon_ai_agent.group_ai_agent_admin 1 1 1 1
32 access_ai_report_template_chart_user 图表配置-用户只读 model_ai_report_template_chart yuthon_ai_agent.group_ai_agent_user 1 0 0 0
33 access_ai_report_template_chart_admin 图表配置-管理员 model_ai_report_template_chart yuthon_ai_agent.group_ai_agent_admin 1 1 1 1
34 access_ai_import_record_user 附件导入记录-用户 model_ai_import_record yuthon_ai_agent.group_ai_agent_user 1 0 1 0
35 access_ai_import_record_admin 附件导入记录-管理员 model_ai_import_record yuthon_ai_agent.group_ai_agent_admin 1 1 1 1
36 access_ai_import_rule_admin 导入规则-管理员 model_ai_import_rule yuthon_ai_agent.group_ai_agent_admin 1 1 1 1
37 access_ai_import_rule_user 导入规则-用户只读 model_ai_import_rule yuthon_ai_agent.group_ai_agent_user 1 0 0 0
@@ -83,14 +83,21 @@ export class AiChatAction extends Component {
onFileSelected(ev) {
const file = ev.target.files[0];
if (!file) return;
const allowed = ["image/png", "image/jpeg", "image/jpg", "image/bmp", "image/gif", "application/pdf"];
if (!allowed.includes(file.type)) {
alert("仅支持 PNG、JPG、BMP、GIF、PDF 文件");
const IMG_TYPES = ["image/png", "image/jpeg", "image/jpg", "image/bmp", "image/gif"];
const EXCEL_TYPES = ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel", "text/csv"];
const ALLOWED = IMG_TYPES.concat(["application/pdf"], EXCEL_TYPES);
const isExcelByName = file.name && /\\.(xlsx|xls|csv)$/i.test(file.name);
if (!ALLOWED.includes(file.type) && !isExcelByName) {
alert("仅支持 PNG、JPG、BMP、GIF、PDF、Excel(.xlsx/.xls/.csv)文件");
ev.target.value = "";
return;
}
if (file.size > 10 * 1024 * 1024) {
alert("文件不能超过 10MB");
const isExcel = EXCEL_TYPES.includes(file.type) || isExcelByName;
const maxSize = isExcel ? 20 * 1024 * 1024 : 10 * 1024 * 1024;
const sizeLabel = isExcel ? "20MB" : "10MB";
if (file.size > maxSize) {
alert("文件不能超过 " + sizeLabel);
ev.target.value = "";
return;
}
+14 -7
View File
@@ -145,14 +145,21 @@ function triggerFileInput() {
function onFileSelected(e) {
const file = e.target.files[0];
if (!file) return;
const allowed = ["image/png", "image/jpeg", "image/jpg", "image/bmp", "image/gif", "application/pdf"];
if (!allowed.includes(file.type)) {
alert("仅支持 PNG、JPG、BMP、GIF、PDF 文件");
const IMG_TYPES = ["image/png", "image/jpeg", "image/jpg", "image/bmp", "image/gif"];
const EXCEL_TYPES = ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/vnd.ms-excel", "text/csv"];
const ALLOWED = IMG_TYPES.concat(["application/pdf"], EXCEL_TYPES);
const isExcelByName = file.name && /\\.(xlsx|xls|csv)$/i.test(file.name);
if (!ALLOWED.includes(file.type) && !isExcelByName) {
alert("仅支持 PNG、JPG、BMP、GIF、PDF、Excel(.xlsx/.xls/.csv)文件");
e.target.value = "";
return;
}
if (file.size > 10 * 1024 * 1024) {
alert("文件不能超过 10MB");
const isExcel = EXCEL_TYPES.includes(file.type) || isExcelByName;
const maxSize = isExcel ? 20 * 1024 * 1024 : 10 * 1024 * 1024;
const sizeLabel = isExcel ? "20MB" : "10MB";
if (file.size > maxSize) {
alert("文件不能超过 " + sizeLabel);
e.target.value = "";
return;
}
@@ -345,11 +352,11 @@ function renderChat() {
<div id="yuthon_attach_preview" class="o_ai_attach_preview" style="display:${showConvList ? 'none' : 'flex'}">
</div>
<div class="o_ai_input_area" style="display:${showConvList ? 'none' : 'flex'}">
<button id="yuthon_attach_btn" class="o_ai_attach_btn" title="上传图片或PDF">
<button id="yuthon_attach_btn" class="o_ai_attach_btn" title="上传图片/PDF/Excel">
<i class="fa fa-paperclip"></i>
</button>
<input type="file" id="yuthon_file_input"
accept="image/png,image/jpeg,image/jpg,image/bmp,image/gif,application/pdf"
accept="image/png,image/jpeg,image/jpg,image/bmp,image/gif,application/pdf,.xlsx,.xls,.csv"
style="display:none"/>
<textarea id="yuthon_chat_input" class="o_ai_input"
placeholder="输入消息... 支持粘贴图片" rows="1"></textarea>
@@ -82,11 +82,11 @@
<!-- 输入区 -->
<div class="ai-chat-input-area p-2 border-top d-flex gap-2 align-items-center">
<button class="btn btn-outline-secondary btn-sm ai-chat-attach-btn"
t-on-click="triggerFileInput" title="上传图片或PDF(支持 Ctrl+V 粘贴图片)">
t-on-click="triggerFileInput" title="上传图片/PDF/Excel(支持 Ctrl+V 粘贴图片)">
<i class="fa fa-paperclip"/>
</button>
<input type="file" id="ai_chat_file_input"
accept="image/png,image/jpeg,image/jpg,image/bmp,image/gif,application/pdf"
accept="image/png,image/jpeg,image/jpg,image/bmp,image/gif,application/pdf,.xlsx,.xls,.csv"
t-on-change="onFileSelected"
style="display:none"/>
<input type="text" class="form-control"
+113 -5
View File
@@ -135,21 +135,37 @@ class WecomAiCallback(http.Controller):
db_name = request.env.cr.dbname
threading.Thread(
target=self._async_process_ai_message,
args=(db_name, wecom_userid, "请帮我看看这张图片里有什么", media_id),
args=(db_name, wecom_userid, "请帮我看看这张图片里有什么", media_id, 'image'),
daemon=True,
).start()
return Response("", content_type='text/plain')
# 文件消息(Excel/CSV):下载 → 解析 → 交给 AI
if msg_type == 'file':
media_id = msg.get('media_id', '')
if not wecom_userid or not media_id:
return Response("", content_type='text/plain')
_logger.info("收到企微文件消息: user=%s, media_id=%s", wecom_userid, media_id[:20])
db_name = request.env.cr.dbname
threading.Thread(
target=self._async_process_ai_message,
args=(db_name, wecom_userid, "请帮我分析这个文件的内容", media_id, 'file'),
daemon=True,
).start()
return Response("", content_type='text/plain')
# 其他类型暂不处理
_logger.info("AI回调收到非文本/图片消息类型: %s,暂不处理", msg_type)
_logger.info("AI回调收到非文本/图片/文件消息类型: %s,暂不处理", msg_type)
return Response("", content_type='text/plain')
def _async_process_ai_message(self, db_name, wecom_userid, content, media_id=None):
def _async_process_ai_message(self, db_name, wecom_userid, content, media_id=None, media_type='image'):
"""
后台线程: 处理 AI 对话并推送回复
使用独立的数据库游标和环境
:param media_id: 可选,图片消息的 media_id。不为空时先下载→OCR→拼入消息
:param media_id: 可选,图片/文件消息的 media_id
:param media_type: 'image' | 'file'
"""
try:
registry = api.Registry(db_name)
@@ -158,9 +174,17 @@ class WecomAiCallback(http.Controller):
attachment_ids = None
if media_id:
if media_type == 'file':
attachment_ids = self._download_and_parse_file(env, wecom_userid, media_id)
else:
attachment_ids = self._download_and_ocr_media(env, wecom_userid, media_id)
if not attachment_ids:
# 下载/OCR失败,给用户一个提示,不再调AI
# 下载/解析失败,给用户一个提示,不再调AI
if media_type == 'file':
env['wecom.ai.chat']._send_wecom_reply(
wecom_userid, "文件已收到,但解析失败。请确保发送的是 .xlsx/.xls/.csv 格式的 Excel 文件。")
else:
env['wecom.ai.chat']._send_wecom_reply(
wecom_userid, "图片已收到,但未能从中提取到文字信息。请尝试发送更清晰的图片(含印刷文字)。")
cr.commit()
@@ -245,3 +269,87 @@ class WecomAiCallback(http.Controller):
except Exception as e:
_logger.exception('[WECOM-OCR] 下载/OCR 异常: %s', e)
return None
def _download_and_parse_file(self, env, wecom_userid, media_id):
"""下载企微文件 → 创建 documents.document → Excel解析 → 返回 [document_id]
支持 .xlsx / .xls / .csv
:return: [int] 或 None
"""
try:
# 1. 下载文件
content, error = env['wecom.apps'].sudo().download_agent_media(media_id)
if content is None:
_logger.error('[WECOM-FILE] 下载失败: %s', error)
return None
_logger.info('[WECOM-FILE] 下载成功: %d bytes', len(content))
# 2. 确定 mimetype(根据魔数)
mimetype = 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet'
ext = '.xlsx'
if content[:2] == b'PK':
mimetype = 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet'
ext = '.xlsx'
elif content[:2] == b'\xd0\xcf':
mimetype = 'application/vnd.ms-excel'
ext = '.xls'
else:
# 尝试按文本解码 → CSV
try:
_ = content.decode('utf-8-sig')
mimetype = 'text/csv'
ext = '.csv'
except:
_logger.warning('[WECOM-FILE] 无法识别文件类型,降级为 xlsx')
filename = f'wecom_file_{media_id[:8]}{ext}'
# 3. 创建 documents.document
DocFolder = env['documents.folder'].sudo()
folder = DocFolder.search([('name', '=', 'AI 聊天附件')], limit=1)
if not folder:
folder = DocFolder.create({'name': 'AI 聊天附件'})
document = env['documents.document'].sudo().create({
'name': filename,
'folder_id': folder.id,
'mimetype': mimetype,
'raw': content,
})
_logger.info('[WECOM-FILE] 文档已创建: id=%s', document.id)
# 4. Excel 解析
try:
from odoo.addons.yuthon_ai_agent.utils.excel_parser import parse_excel_to_markdown
except ImportError:
_logger.warning('[WECOM-FILE] 无法导入 Excel 解析器')
return [document.id]
markdown_text, meta = parse_excel_to_markdown(content, filename)
if meta.get('error'):
_logger.error('[WECOM-FILE] 解析失败: %s', meta['error'])
return [document.id]
document.write({'ai_ocr_content': markdown_text})
_logger.info('[WECOM-FILE] ✅ 解析成功: %d字符 | sheets=%s | rows=%s',
len(markdown_text), meta.get('sheets'), meta.get('total_rows'))
# 5. 创建导入记录
try:
env['ai.import.record'].sudo().create({
'document_id': document.id,
'filename': filename,
'file_type': 'excel',
'user_id': env.ref('base.user_root').id,
'wecom_userid': wecom_userid,
'ocr_status': 'success' if not meta.get('truncated') else 'success',
'ocr_content': markdown_text,
})
except Exception as e:
_logger.warning('[WECOM-FILE] 创建导入记录失败: %s', e)
return [document.id]
except Exception as e:
_logger.exception('[WECOM-FILE] 下载/解析异常: %s', e)
return None