excel识别

This commit is contained in:
李鹏宇
2026-06-12 18:11:26 +08:00
parent 899e4c62ff
commit d4252e0a76
4 changed files with 718 additions and 0 deletions
+275
View File
@@ -0,0 +1,275 @@
# -*- coding: utf-8 -*-
"""
AI 附件导入记录模型
记录每次 Excel/图片 附件的识别结果,支持后台查看和重新导入
"""
from odoo import models, fields, api
from odoo import tools
import logging
_logger = logging.getLogger(__name__)
class AiImportRecord(models.Model):
"""
AI 附件导入记录
每次用户上传附件(图片/Excel)并触发 AI 识别后,创建一条记录。
用于后台审计、重新导入、数据分析。
"""
_name = 'ai.import.record'
_description = 'AI 附件导入记录'
_order = 'create_date desc'
_rec_name = 'name'
name = fields.Char(string='名称', compute='_compute_name', store=True)
create_date = fields.Datetime(string='创建时间', readonly=True)
create_uid = fields.Many2one('res.users', string='创建人', readonly=True)
# 关联信息
conversation_id = fields.Many2one('ai.conversation', string='对话', ondelete='set null', index=True)
message_id = fields.Many2one('ai.message', string='消息', ondelete='set null', index=True)
user_id = fields.Many2one('res.users', string='用户', index=True)
wecom_userid = fields.Char(string='企微UserID', index=True)
# 附件信息
document_id = fields.Many2one('documents.document', string='原始附件', ondelete='set null', index=True)
filename = fields.Char(string='文件名')
file_type = fields.Selection([
('image', '图片'),
('excel', 'Excel'),
('csv', 'CSV'),
('pdf', 'PDF'),
('other', '其他'),
], string='文件类型', default='other', index=True)
# 识别结果
ocr_content = fields.Text(string='OCR/解析内容', help='图片OCR文字或Excel解析后的Markdown文本')
ocr_status = fields.Selection([
('pending', '待处理'),
('success', '成功'),
('no_text', '无文字'),
('failed', '失败'),
], string='识别状态', default='pending', index=True)
ocr_error = fields.Text(string='识别错误')
# 导入信息
import_status = fields.Selection([
('none', '未导入'),
('pending', '待导入'),
('success', '导入成功'),
('failed', '导入失败'),
('skipped', '已跳过'),
], string='导入状态', default='none', index=True)
import_rule_id = fields.Many2one('ai.import.rule', string='使用的导入规则', ondelete='set null')
import_result = fields.Text(string='导入结果', help='导入成功/失败的详细信息')
imported_record_ids = fields.Text(string='已导入记录ID', help='JSON格式,记录导入的Odoo记录ID列表')
# AI 回复
ai_reply = fields.Text(string='AI回复内容', help='AI对该附件的回复')
@api.depends('filename', 'file_type', 'create_date')
def _compute_name(self):
for rec in self:
prefix = dict(self._fields['file_type'].selection).get(rec.file_type, '文件')
time_str = rec.create_date.strftime('%m-%d %H:%M') if rec.create_date else ''
rec.name = u'{prefix} - {rec.filename or "未命名"} - {time_str}'
def action_reparse(self):
"""重新解析附件(管理员手动触发)"""
self.ensure_one()
if not self.document_id:
return {'warning': u'原始附件已删除,无法重新解析'}
if self.file_type in ('excel', 'csv'):
from ..utils.excel_parser import parse_excel_to_markdown
file_bytes = self.document_id.raw
if not file_bytes:
self.write({'ocr_status': 'failed', 'ocr_error': u'无法读取文件内容'})
return True
markdown_text, meta = parse_excel_to_markdown(file_bytes, self.filename or '')
if meta.get('error'):
self.write({'ocr_status': 'failed', 'ocr_error': meta['error']})
else:
self.write({
'ocr_status': 'success',
'ocr_content': markdown_text,
'ocr_error': '',
})
elif self.file_type == 'image':
# 重新OCR
provider = self.env['ai.provider'].search([('active', '=', True)], limit=1)
if not provider or not provider.ocr_enabled:
self.write({'ocr_status': 'failed', 'ocr_error': u'OCR未启用'})
return True
from ..models.ocr_provider import get_ocr_instance
ocr, init_err = get_ocr_instance(provider)
if ocr is None:
self.write({'ocr_status': 'failed', 'ocr_error': init_err or u'OCR实例创建失败'})
return True
file_bytes = self.document_id.raw
mimetype = self.document_id.mimetype or 'image/png'
result = ocr.recognize_general(file_bytes, mimetype)
if result.get('success'):
ocr_text = result.get('text') or ''
if ocr_text:
self.write({
'ocr_status': 'success',
'ocr_content': ocr_text,
'ocr_error': '',
})
else:
self.write({'ocr_status': 'no_text', 'ocr_error': u'图片中未识别到文字'})
else:
self.write({'ocr_status': 'failed', 'ocr_error': result.get('error', u'OCR调用失败')})
return True
def action_view_document(self):
"""打开原始附件"""
self.ensure_one()
if not self.document_id:
return {}
return {
'type': 'ir.actions.act_window',
'name': u'原始附件',
'res_model': 'documents.document',
'res_id': self.document_id.id,
'view_mode': 'form',
'target': 'new',
}
class AiImportRule(models.Model):
"""
AI 附件导入规则
定义"某种类型的附件应该如何解析和导入"。
例如:
- Excel 花名册 → 导入 hr.employee
- Excel 供应商列表 → 导入 res.partner
- 图片发票 → 调用腾讯云发票OCR → 导入 account.move
"""
_name = 'ai.import.rule'
_description = u'AI 附件导入规则'
_order = 'sequence, id'
name = fields.Char(string='规则名称', required=True)
sequence = fields.Integer(string='优先级', default=10)
active = fields.Boolean(string='启用', default=True)
# 匹配条件
file_type = fields.Selection([
('image', u'图片'),
('excel', u'Excel/CSV'),
('any', u'任意'),
], string='文件类型', default='any', required=True)
filename_pattern = fields.Char(
string='文件名模式',
help=u'文件名匹配正则表达式,如 "花名册.*xlsx" 表示花名册开头的xlsx文件'
)
mime_type_pattern = fields.Char(
string='MIME类型模式',
help=u'MIME类型匹配,如 "image/*" 或 "application/vnd.openxmlformats*"'
)
# 解析配置
parser_type = fields.Selection([
('auto', u'自动(图片OCR,Excel解析为Markdown)'),
('excel_markdown', u'Excel→Markdown'),
('excel_json', u'Excel→JSON(按模板列映射)'),
('image_ocr', u'图片→OCR文字'),
('image_invoice_ocr', u'图片→腾讯云发票OCR'),
('image_idcard_ocr', u'图片→腾讯云身份证OCR'),
('custom', u'自定义解析器(Python代码)'),
], string='解析方式', default='auto', required=True)
# 导入配置
target_model = fields.Char(
string='目标模型',
help=u'导入数据的Odoo模型,如 hr.employee、res.partner'
)
field_mapping = fields.Text(
string='字段映射',
help=u'JSON格式,定义Excel列名→Odoo字段的映射。如 {"姓名": "name", "手机号": "mobile"}'
)
import_mode = fields.Selection([
('create', u'仅创建'),
('update', u'仅更新(按唯一键匹配)'),
('create_update', u'创建或更新'),
], string='导入模式', default='create')
# 唯一键(用于 update 模式匹配已有记录)
unique_key = fields.Char(
string='唯一键字段',
help=u'用于匹配已有记录的字段,如 "id_card" 或 "name,mobile"'
)
# AI 提示词(告诉AI如何解读这个附件)
ai_prompt = fields.Text(
string='AI提示词',
help=u'告诉AI这个附件的用途和解读方式。如:"这是员工花名册,请统计各部门人数"'
)
# 统计
match_count = fields.Integer(string='匹配次数', compute='_compute_match_count')
def _compute_match_count(self):
for rec in self:
rec.match_count = self.env['ai.import.record'].search_count([
('import_rule_id', '=', rec.id)
])
def test_match(self, filename, file_type, mime_type=''):
"""
测试文件是否匹配此规则
:return: True / False
"""
self.ensure_one()
# 文件类型匹配
if self.file_type != 'any' and self.file_type != file_type:
return False
# 文件名模式匹配
if self.filename_pattern:
import re
try:
if not re.search(self.filename_pattern, filename):
return False
except re.error:
_logger.warning('[ImportRule] 文件名正则无效: %s', self.filename_pattern)
return False
# MIME类型匹配
if self.mime_type_pattern:
import fnmatch
if not fnmatch.fnmatch(mime_type, self.mime_type_pattern):
return False
return True
@api.model
def find_matching_rule(self, filename, file_type, mime_type=''):
"""
根据文件名、类型找到最适合的导入规则(按优先级)
:return: ai.import.rule 记录或 False
"""
rules = self.search([('active', '=', True)], order='sequence, id')
for rule in rules:
if rule.test_match(filename, file_type, mime_type):
_logger.info('[ImportRule] 匹配规则: %s (file=%s, type=%s)', rule.name, filename, file_type)
return rule
return False
def action_view_matched_records(self):
"""查看使用该规则的历史记录"""
self.ensure_one()
return {
'type': 'ir.actions.act_window',
'name': u'%s - 历史记录' % self.name,
'res_model': 'ai.import.record',
'view_mode': 'tree,form',
'domain': [('import_rule_id', '=', self.id)],
'context': {'default_import_rule_id': self.id},
}
+2
View File
@@ -0,0 +1,2 @@
# -*- coding: utf-8 -*-
from . import excel_parser
+234
View File
@@ -0,0 +1,234 @@
# -*- coding: utf-8 -*-
"""
Excel 解析工具:将 Excel 文件转为 AI 可读的 Markdown 文本
支持 .xlsx / .xls / .csv
"""
import io
import logging
from typing import Tuple, List, Dict, Any
_logger = logging.getLogger(__name__)
# 默认限制
MAX_ROWS = 500 # 单 Sheet 最大行数(超出的给统计摘要)
MAX_TOTAL_ROWS = 2000 # 所有 Sheet 总行数上限
MAX_FILE_SIZE_MB = 20 # 文件大小上限(MB)
def parse_excel_to_markdown(
file_bytes: bytes,
filename: str,
max_rows: int = MAX_ROWS,
max_total_rows: int = MAX_TOTAL_ROWS,
) -> Tuple[str, Dict[str, Any]]:
"""
将 Excel/CSV 解析为 Markdown 文本,供 AI 阅读。
:param file_bytes: 文件二进制内容
:param filename: 文件名(用于判断类型)
:param max_rows: 单 Sheet 最大行数
:param max_total_rows: 所有 Sheet 总行数上限
:return: (markdown_text, meta_dict)
meta_dict: {sheets, total_rows, truncated, file_type, error}
"""
meta = {
'sheets': [],
'total_rows': 0,
'truncated': False,
'file_type': '',
'error': '',
}
try:
import pandas as pd
except ImportError:
meta['error'] = 'pandas 未安装,无法解析 Excel 文件。请运行:pip install pandas openpyxl'
_logger.error('[ExcelParser] pandas 未安装')
return '', meta
file_type = _detect_file_type(file_bytes, filename)
meta['file_type'] = file_type
try:
if file_type == 'csv':
return _parse_csv(file_bytes, max_rows, max_total_rows, meta)
elif file_type in ('xlsx', 'xls'):
return _parse_excel(file_bytes, max_rows, max_total_rows, meta)
else:
meta['error'] = f'不支持的文件类型:{file_type}'
return '', meta
except Exception as e:
_logger.exception('[ExcelParser] 解析失败: %s', e)
meta['error'] = f'解析失败:{e}'
return '', meta
def _detect_file_type(file_bytes: bytes, filename: str) -> str:
"""根据文件内容和扩展名判断类型"""
# 1. 优先看扩展名
lower_name = filename.lower()
if lower_name.endswith('.csv'):
return 'csv'
if lower_name.endswith('.xlsx'):
return 'xlsx'
if lower_name.endswith('.xls'):
return 'xls'
# 2. 看魔数(文件头)
if file_bytes[:2] == b'PK': # ZIP (xlsx)
return 'xlsx'
if file_bytes[:2] == b'\xd0\xcf': # Old Excel (xls)
return 'xls'
# 尝试当作 CSV
try:
file_bytes.decode('utf-8-sig')
return 'csv'
except:
pass
return 'unknown'
def _parse_csv(bytes_data: bytes, max_rows: int, max_total: int, meta: dict) -> Tuple[str, dict]:
"""解析 CSV 文件"""
import pandas as pd
# 尝试多种编码
encoding = _detect_encoding(bytes_data)
df = pd.read_csv(io.BytesIO(bytes_data), encoding=encoding, nrows=max_total + 1)
return _df_to_markdown(df, 'CSV', max_rows, max_total, meta)
def _parse_excel(bytes_data: bytes, max_rows: int, max_total: int, meta: dict) -> Tuple[str, dict]:
"""解析 Excel 文件(支持多 Sheet)"""
import pandas as pd
xl = pd.ExcelFile(io.BytesIO(bytes_data))
sheet_names = xl.sheet_names
meta['sheets'] = sheet_names
parts = []
total_rows = 0
for sheet_name in sheet_names:
df = pd.read_excel(xl, sheet_name=sheet_name, nrows=max_total + 1 - total_rows)
# 检查是否超出总行数
if total_rows + len(df) > max_total:
df = df.head(max_total - total_rows)
meta['truncated'] = True
_logger.warning('[ExcelParser] 总行数超过 %s,已截断', max_total)
sheet_md, sheet_meta = _df_to_markdown(df, sheet_name, max_rows, max_total, meta, is_single_sheet=(len(sheet_names) == 1))
parts.append(sheet_md)
total_rows += len(df)
if meta.get('truncated'):
break
meta['total_rows'] = total_rows
return '\n\n'.join(parts), meta
def _df_to_markdown(
df,
sheet_name: str,
max_rows: int,
max_total: int,
meta: dict,
is_single_sheet: bool = False,
) -> Tuple[str, dict]:
"""
将 DataFrame 转为 Markdown。
策略:
- 行数 <= max_rows:完整输出
- 行数 > max_rows:输出统计摘要 + 前 max_rows 行
"""
row_count = len(df)
col_count = len(df.columns)
if not is_single_sheet:
parts = [f'## Sheet: {sheet_name}']
else:
parts = []
parts.append(f'({row_count} 行 × {col_count} 列)\n')
if row_count == 0:
parts.append('(空表)')
return '\n'.join(parts), meta
# 统计摘要(数值列)
numeric_cols = df.select_dtypes(include=['number']).columns.tolist()
if numeric_cols:
parts.append('【数值列统计】')
for col in numeric_cols[:10]: # 最多10个数值列
series = df[col].dropna()
if len(series) > 0:
parts.append(f' - {col}:最小={series.min():.2f},最大={series.max():.2f},平均={series.mean():.2f},合计={series.sum():.2f}')
parts.append('')
# 分类列(唯一值数量)
cat_cols = [c for c in df.columns if c not in numeric_cols]
if cat_cols:
parts.append('【分类列概况】')
for col in cat_cols[:5]: # 最多5个分类列
unique_count = df[col].nunique()
top_val = df[col].mode().iloc[0] if len(df[col].mode()) > 0 else '(无)'
parts.append(f' - {col}:共 {unique_count} 种值,最常见「{top_val}」')
parts.append('')
# 数据内容
if row_count <= max_rows:
parts.append('【完整数据】')
parts.append(df.to_markdown(index=False))
else:
parts.append(f'【前 {max_rows} 行数据】(共 {row_count} 行,已截断)')
parts.append(df.head(max_rows).to_markdown(index=False))
meta['truncated'] = True
return '\n'.join(parts), meta
def _detect_encoding(bytes_data: bytes) -> str:
"""检测 CSV 文件编码"""
for enc in ['utf-8-sig', 'utf-8', 'gbk', 'gb2312', 'iso-8859-1']:
try:
bytes_data.decode(enc)
return enc
except:
continue
return 'utf-8'
def validate_excel_file(file_bytes: bytes, filename: str) -> Dict[str, Any]:
"""
验证 Excel 文件是否合法、大小是否超限。
返回:{valid, error, file_type, size_mb}
"""
size_mb = len(file_bytes) / 1024 / 1024
if size_mb > MAX_FILE_SIZE_MB:
return {
'valid': False,
'error': f'文件过大({size_mb:.1f}MB),上限 {MAX_FILE_SIZE_MB}MB',
'file_type': '',
'size_mb': round(size_mb, 1),
}
file_type = _detect_file_type(file_bytes, filename)
if file_type == 'unknown':
return {
'valid': False,
'error': f'不支持的文件格式,仅支持 .xlsx / .xls / .csv',
'file_type': file_type,
'size_mb': round(size_mb, 1),
}
return {
'valid': True,
'error': '',
'file_type': file_type,
'size_mb': round(size_mb, 1),
}
@@ -0,0 +1,207 @@
<?xml version="1.0" encoding="utf-8"?>
<odoo>
<!-- ========== 附件导入记录 ========== -->
<record id="view_ai_import_record_tree" model="ir.ui.view">
<field name="name">ai.import.record.tree</field>
<field name="model">ai.import.record</field>
<field name="arch" type="xml">
<tree editable="bottom" decoration-warning="ocr_status == 'failed'" decoration-muted="import_status == 'skipped'">
<field name="create_date" string="时间" optional="hide"/>
<field name="filename" string="文件名"/>
<field name="file_type" string="类型"/>
<field name="ocr_status" string="识别状态" widget="badge"
decoration-success="ocr_status == 'success'"
decoration-warning="ocr_status == 'failed'"
decoration-info="ocr_status == 'pending'"
decoration-muted="ocr_status == 'no_text'"/>
<field name="import_status" string="导入状态" widget="badge"
decoration-success="import_status == 'success'"
decoration-warning="import_status == 'failed'"
decoration-info="import_status == 'pending'"
decoration-muted="import_status == 'skipped'"/>
<field name="create_uid" string="用户" optional="hide"/>
<field name="wecom_userid" string="企微ID" optional="hide"/>
<field name="import_rule_id" string="导入规则" optional="hide"/>
</tree>
</field>
</record>
<record id="view_ai_import_record_form" model="ir.ui.view">
<field name="name">ai.import.record.form</field>
<field name="model">ai.import.record</field>
<field name="arch" type="xml">
<form>
<header>
<button name="action_reparse" string="重新解析" type="object"
class="btn-secondary" confirm="确定重新解析此附件?"/>
<button name="action_view_document" string="查看原始附件" type="object"
class="btn-secondary"/>
</header>
<sheet>
<div class="oe_title">
<h1><field name="name"/></h1>
</div>
<group>
<group string="基本信息">
<field name="filename"/>
<field name="file_type"/>
<field name="create_date"/>
<field name="create_uid"/>
<field name="wecom_userid"/>
</group>
<group string="识别信息">
<field name="ocr_status" widget="badge"
decoration-success="ocr_status == 'success'"
decoration-warning="ocr_status == 'failed'"
decoration-info="ocr_status == 'pending'"/>
<field name="ocr_error" attrs="{'invisible': [('ocr_error', '=', '')]}"/>
<field name="import_status" widget="badge"/>
<field name="import_rule_id"/>
</group>
</group>
<group string="关联">
<field name="user_id"/>
<field name="conversation_id"/>
<field name="document_id"/>
</group>
<notebook>
<page string="识别结果">
<field name="ocr_content" nolabel="1" readonly="1"
widget="html" class="oe_read_only"/>
</page>
<page string="AI回复">
<field name="ai_reply" nolabel="1" readonly="1"
widget="html"/>
</page>
<page string="导入结果">
<field name="import_result" nolabel="1"/>
<field name="imported_record_ids" nolabel="1"/>
</page>
</notebook>
</sheet>
</form>
</field>
</record>
<record id="view_ai_import_record_search" model="ir.ui.view">
<field name="name">ai.import.record.search</field>
<field name="model">ai.import.record</field>
<field name="arch" type="xml">
<search>
<field name="filename"/>
<field name="ocr_content"/>
<filter name="status_success" string="识别成功"
domain="[('ocr_status', '=', 'success')]"/>
<filter name="status_failed" string="识别失败"
domain="[('ocr_status', 'in', ['failed', 'no_text'])]"/>
<separator/>
<filter name="file_image" string="图片"
domain="[('file_type', '=', 'image')]"/>
<filter name="file_excel" string="Excel"
domain="[('file_type', 'in', ['excel', 'csv'])]"/>
<separator/>
<filter name="my_imports" string="我的导入"
domain="[('user_id', '=', uid)]"/>
<group expand="1" string="分组">
<filter name="group_file_type" string="按文件类型"
context="{'group_by': 'file_type'}"/>
<filter name="group_status" string="按识别状态"
context="{'group_by': 'ocr_status'}"/>
<filter name="group_user" string="按用户"
context="{'group_by': 'user_id'}"/>
</group>
</search>
</field>
</record>
<record id="action_ai_import_record" model="ir.actions.act_window">
<field name="name">附件导入记录</field>
<field name="res_model">ai.import.record</field>
<field name="view_mode">tree,form</field>
<field name="search_view_id" ref="view_ai_import_record_search"/>
</record>
<menuitem id="menu_ai_import_record"
name="附件导入记录"
parent="yuthon_ai_agent.menu_ai_root"
action="action_ai_import_record"
sequence="50"
groups="yuthon_ai_agent.group_ai_agent_admin"/>
<!-- ========== 导入规则 ========== -->
<record id="view_ai_import_rule_tree" model="ir.ui.view">
<field name="name">ai.import.rule.tree</field>
<field name="model">ai.import.rule</field>
<field name="arch" type="xml">
<tree editable="bottom">
<field name="sequence" widget="handle"/>
<field name="name"/>
<field name="file_type"/>
<field name="parser_type"/>
<field name="target_model"/>
<field name="match_count" string="匹配次数"/>
<field name="active"/>
</tree>
</field>
</record>
<record id="view_ai_import_rule_form" model="ir.ui.view">
<field name="name">ai.import.rule.form</field>
<field name="model">ai.import.rule</field>
<field name="arch" type="xml">
<form>
<header>
<button name="action_view_matched_records" string="查看匹配记录"
type="object" class="btn-secondary"/>
</header>
<sheet>
<div class="oe_title">
<h1><field name="name"/></h1>
</div>
<group>
<group string="基本配置">
<field name="sequence"/>
<field name="active"/>
<field name="file_type"/>
</group>
<group string="匹配条件">
<field name="filename_pattern" placeholder="如:花名册.*xlsx"/>
<field name="mime_type_pattern" placeholder="如:application/vnd.openxmlformats*"/>
</group>
</group>
<group string="解析与导入配置">
<field name="parser_type"/>
<field name="target_model" placeholder="如:hr.employee"/>
<field name="import_mode"/>
<field name="unique_key" placeholder="如:id_card"
attrs="{'invisible': [('import_mode', '=', 'create')]}"/>
</group>
<group string="字段映射(JSON格式)">
<field name="field_mapping" nolabel="1" widget="ace"
options="{'mode': 'json'}"
placeholder='{"Excel列名": "odoo字段名"} 范例:{"姓名": "name", "手机号": "mobile_phone"}'/>
</group>
<group string="AI提示词">
<field name="ai_prompt" nolabel="1"
placeholder="告诉AI这个附件的用途和解读方式"/>
</group>
</sheet>
</form>
</field>
</record>
<record id="action_ai_import_rule" model="ir.actions.act_window">
<field name="name">导入规则配置</field>
<field name="res_model">ai.import.rule</field>
<field name="view_mode">tree,form</field>
</record>
<menuitem id="menu_ai_import_rule"
name="导入规则"
parent="yuthon_ai_agent.menu_ai_root"
action="action_ai_import_rule"
sequence="55"
groups="yuthon_ai_agent.group_ai_agent_admin"/>
</odoo>