From d4252e0a7613dc10fc63ab120f9761c0ba4e9532 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E6=9D=8E=E9=B9=8F=E5=AE=87?= Date: Fri, 12 Jun 2026 18:11:26 +0800 Subject: [PATCH] =?UTF-8?q?excel=E8=AF=86=E5=88=AB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- yuthon_ai_agent/models/ai_import_record.py | 275 ++++++++++++++++++ yuthon_ai_agent/utils/__init__.py | 2 + yuthon_ai_agent/utils/excel_parser.py | 234 +++++++++++++++ .../views/ai_import_record_views.xml | 207 +++++++++++++ 4 files changed, 718 insertions(+) create mode 100644 yuthon_ai_agent/models/ai_import_record.py create mode 100644 yuthon_ai_agent/utils/__init__.py create mode 100644 yuthon_ai_agent/utils/excel_parser.py create mode 100644 yuthon_ai_agent/views/ai_import_record_views.xml diff --git a/yuthon_ai_agent/models/ai_import_record.py b/yuthon_ai_agent/models/ai_import_record.py new file mode 100644 index 000000000..eab635df0 --- /dev/null +++ b/yuthon_ai_agent/models/ai_import_record.py @@ -0,0 +1,275 @@ +# -*- coding: utf-8 -*- +""" +AI 附件导入记录模型 +记录每次 Excel/图片 附件的识别结果,支持后台查看和重新导入 +""" +from odoo import models, fields, api +from odoo import tools +import logging + +_logger = logging.getLogger(__name__) + + +class AiImportRecord(models.Model): + """ + AI 附件导入记录 + 每次用户上传附件(图片/Excel)并触发 AI 识别后,创建一条记录。 + 用于后台审计、重新导入、数据分析。 + """ + _name = 'ai.import.record' + _description = 'AI 附件导入记录' + _order = 'create_date desc' + _rec_name = 'name' + + name = fields.Char(string='名称', compute='_compute_name', store=True) + create_date = fields.Datetime(string='创建时间', readonly=True) + create_uid = fields.Many2one('res.users', string='创建人', readonly=True) + + # 关联信息 + conversation_id = fields.Many2one('ai.conversation', string='对话', ondelete='set null', index=True) + message_id = fields.Many2one('ai.message', string='消息', ondelete='set null', index=True) + user_id = fields.Many2one('res.users', string='用户', index=True) + wecom_userid = fields.Char(string='企微UserID', index=True) + + # 附件信息 + document_id = fields.Many2one('documents.document', string='原始附件', ondelete='set null', index=True) + filename = fields.Char(string='文件名') + file_type = fields.Selection([ + ('image', '图片'), + ('excel', 'Excel'), + ('csv', 'CSV'), + ('pdf', 'PDF'), + ('other', '其他'), + ], string='文件类型', default='other', index=True) + + # 识别结果 + ocr_content = fields.Text(string='OCR/解析内容', help='图片OCR文字或Excel解析后的Markdown文本') + ocr_status = fields.Selection([ + ('pending', '待处理'), + ('success', '成功'), + ('no_text', '无文字'), + ('failed', '失败'), + ], string='识别状态', default='pending', index=True) + ocr_error = fields.Text(string='识别错误') + + # 导入信息 + import_status = fields.Selection([ + ('none', '未导入'), + ('pending', '待导入'), + ('success', '导入成功'), + ('failed', '导入失败'), + ('skipped', '已跳过'), + ], string='导入状态', default='none', index=True) + import_rule_id = fields.Many2one('ai.import.rule', string='使用的导入规则', ondelete='set null') + import_result = fields.Text(string='导入结果', help='导入成功/失败的详细信息') + imported_record_ids = fields.Text(string='已导入记录ID', help='JSON格式,记录导入的Odoo记录ID列表') + + # AI 回复 + ai_reply = fields.Text(string='AI回复内容', help='AI对该附件的回复') + + @api.depends('filename', 'file_type', 'create_date') + def _compute_name(self): + for rec in self: + prefix = dict(self._fields['file_type'].selection).get(rec.file_type, '文件') + time_str = rec.create_date.strftime('%m-%d %H:%M') if rec.create_date else '' + rec.name = u'{prefix} - {rec.filename or "未命名"} - {time_str}' + + def action_reparse(self): + """重新解析附件(管理员手动触发)""" + self.ensure_one() + if not self.document_id: + return {'warning': u'原始附件已删除,无法重新解析'} + + if self.file_type in ('excel', 'csv'): + from ..utils.excel_parser import parse_excel_to_markdown + file_bytes = self.document_id.raw + if not file_bytes: + self.write({'ocr_status': 'failed', 'ocr_error': u'无法读取文件内容'}) + return True + + markdown_text, meta = parse_excel_to_markdown(file_bytes, self.filename or '') + if meta.get('error'): + self.write({'ocr_status': 'failed', 'ocr_error': meta['error']}) + else: + self.write({ + 'ocr_status': 'success', + 'ocr_content': markdown_text, + 'ocr_error': '', + }) + elif self.file_type == 'image': + # 重新OCR + provider = self.env['ai.provider'].search([('active', '=', True)], limit=1) + if not provider or not provider.ocr_enabled: + self.write({'ocr_status': 'failed', 'ocr_error': u'OCR未启用'}) + return True + + from ..models.ocr_provider import get_ocr_instance + ocr, init_err = get_ocr_instance(provider) + if ocr is None: + self.write({'ocr_status': 'failed', 'ocr_error': init_err or u'OCR实例创建失败'}) + return True + + file_bytes = self.document_id.raw + mimetype = self.document_id.mimetype or 'image/png' + result = ocr.recognize_general(file_bytes, mimetype) + if result.get('success'): + ocr_text = result.get('text') or '' + if ocr_text: + self.write({ + 'ocr_status': 'success', + 'ocr_content': ocr_text, + 'ocr_error': '', + }) + else: + self.write({'ocr_status': 'no_text', 'ocr_error': u'图片中未识别到文字'}) + else: + self.write({'ocr_status': 'failed', 'ocr_error': result.get('error', u'OCR调用失败')}) + + return True + + def action_view_document(self): + """打开原始附件""" + self.ensure_one() + if not self.document_id: + return {} + return { + 'type': 'ir.actions.act_window', + 'name': u'原始附件', + 'res_model': 'documents.document', + 'res_id': self.document_id.id, + 'view_mode': 'form', + 'target': 'new', + } + + +class AiImportRule(models.Model): + """ + AI 附件导入规则 + 定义"某种类型的附件应该如何解析和导入"。 + 例如: + - Excel 花名册 → 导入 hr.employee + - Excel 供应商列表 → 导入 res.partner + - 图片发票 → 调用腾讯云发票OCR → 导入 account.move + """ + _name = 'ai.import.rule' + _description = u'AI 附件导入规则' + _order = 'sequence, id' + + name = fields.Char(string='规则名称', required=True) + sequence = fields.Integer(string='优先级', default=10) + active = fields.Boolean(string='启用', default=True) + + # 匹配条件 + file_type = fields.Selection([ + ('image', u'图片'), + ('excel', u'Excel/CSV'), + ('any', u'任意'), + ], string='文件类型', default='any', required=True) + filename_pattern = fields.Char( + string='文件名模式', + help=u'文件名匹配正则表达式,如 "花名册.*xlsx" 表示花名册开头的xlsx文件' + ) + mime_type_pattern = fields.Char( + string='MIME类型模式', + help=u'MIME类型匹配,如 "image/*" 或 "application/vnd.openxmlformats*"' + ) + + # 解析配置 + parser_type = fields.Selection([ + ('auto', u'自动(图片OCR,Excel解析为Markdown)'), + ('excel_markdown', u'Excel→Markdown'), + ('excel_json', u'Excel→JSON(按模板列映射)'), + ('image_ocr', u'图片→OCR文字'), + ('image_invoice_ocr', u'图片→腾讯云发票OCR'), + ('image_idcard_ocr', u'图片→腾讯云身份证OCR'), + ('custom', u'自定义解析器(Python代码)'), + ], string='解析方式', default='auto', required=True) + + # 导入配置 + target_model = fields.Char( + string='目标模型', + help=u'导入数据的Odoo模型,如 hr.employee、res.partner' + ) + field_mapping = fields.Text( + string='字段映射', + help=u'JSON格式,定义Excel列名→Odoo字段的映射。如 {"姓名": "name", "手机号": "mobile"}' + ) + import_mode = fields.Selection([ + ('create', u'仅创建'), + ('update', u'仅更新(按唯一键匹配)'), + ('create_update', u'创建或更新'), + ], string='导入模式', default='create') + + # 唯一键(用于 update 模式匹配已有记录) + unique_key = fields.Char( + string='唯一键字段', + help=u'用于匹配已有记录的字段,如 "id_card" 或 "name,mobile"' + ) + + # AI 提示词(告诉AI如何解读这个附件) + ai_prompt = fields.Text( + string='AI提示词', + help=u'告诉AI这个附件的用途和解读方式。如:"这是员工花名册,请统计各部门人数"' + ) + + # 统计 + match_count = fields.Integer(string='匹配次数', compute='_compute_match_count') + + def _compute_match_count(self): + for rec in self: + rec.match_count = self.env['ai.import.record'].search_count([ + ('import_rule_id', '=', rec.id) + ]) + + def test_match(self, filename, file_type, mime_type=''): + """ + 测试文件是否匹配此规则 + :return: True / False + """ + self.ensure_one() + # 文件类型匹配 + if self.file_type != 'any' and self.file_type != file_type: + return False + + # 文件名模式匹配 + if self.filename_pattern: + import re + try: + if not re.search(self.filename_pattern, filename): + return False + except re.error: + _logger.warning('[ImportRule] 文件名正则无效: %s', self.filename_pattern) + return False + + # MIME类型匹配 + if self.mime_type_pattern: + import fnmatch + if not fnmatch.fnmatch(mime_type, self.mime_type_pattern): + return False + + return True + + @api.model + def find_matching_rule(self, filename, file_type, mime_type=''): + """ + 根据文件名、类型找到最适合的导入规则(按优先级) + :return: ai.import.rule 记录或 False + """ + rules = self.search([('active', '=', True)], order='sequence, id') + for rule in rules: + if rule.test_match(filename, file_type, mime_type): + _logger.info('[ImportRule] 匹配规则: %s (file=%s, type=%s)', rule.name, filename, file_type) + return rule + return False + + def action_view_matched_records(self): + """查看使用该规则的历史记录""" + self.ensure_one() + return { + 'type': 'ir.actions.act_window', + 'name': u'%s - 历史记录' % self.name, + 'res_model': 'ai.import.record', + 'view_mode': 'tree,form', + 'domain': [('import_rule_id', '=', self.id)], + 'context': {'default_import_rule_id': self.id}, + } diff --git a/yuthon_ai_agent/utils/__init__.py b/yuthon_ai_agent/utils/__init__.py new file mode 100644 index 000000000..2335431b8 --- /dev/null +++ b/yuthon_ai_agent/utils/__init__.py @@ -0,0 +1,2 @@ +# -*- coding: utf-8 -*- +from . import excel_parser diff --git a/yuthon_ai_agent/utils/excel_parser.py b/yuthon_ai_agent/utils/excel_parser.py new file mode 100644 index 000000000..ef0f6383f --- /dev/null +++ b/yuthon_ai_agent/utils/excel_parser.py @@ -0,0 +1,234 @@ +# -*- coding: utf-8 -*- +""" +Excel 解析工具:将 Excel 文件转为 AI 可读的 Markdown 文本 +支持 .xlsx / .xls / .csv +""" +import io +import logging +from typing import Tuple, List, Dict, Any + +_logger = logging.getLogger(__name__) + +# 默认限制 +MAX_ROWS = 500 # 单 Sheet 最大行数(超出的给统计摘要) +MAX_TOTAL_ROWS = 2000 # 所有 Sheet 总行数上限 +MAX_FILE_SIZE_MB = 20 # 文件大小上限(MB) + + +def parse_excel_to_markdown( + file_bytes: bytes, + filename: str, + max_rows: int = MAX_ROWS, + max_total_rows: int = MAX_TOTAL_ROWS, +) -> Tuple[str, Dict[str, Any]]: + """ + 将 Excel/CSV 解析为 Markdown 文本,供 AI 阅读。 + + :param file_bytes: 文件二进制内容 + :param filename: 文件名(用于判断类型) + :param max_rows: 单 Sheet 最大行数 + :param max_total_rows: 所有 Sheet 总行数上限 + :return: (markdown_text, meta_dict) + meta_dict: {sheets, total_rows, truncated, file_type, error} + """ + meta = { + 'sheets': [], + 'total_rows': 0, + 'truncated': False, + 'file_type': '', + 'error': '', + } + + try: + import pandas as pd + except ImportError: + meta['error'] = 'pandas 未安装,无法解析 Excel 文件。请运行:pip install pandas openpyxl' + _logger.error('[ExcelParser] pandas 未安装') + return '', meta + + file_type = _detect_file_type(file_bytes, filename) + meta['file_type'] = file_type + + try: + if file_type == 'csv': + return _parse_csv(file_bytes, max_rows, max_total_rows, meta) + elif file_type in ('xlsx', 'xls'): + return _parse_excel(file_bytes, max_rows, max_total_rows, meta) + else: + meta['error'] = f'不支持的文件类型:{file_type}' + return '', meta + except Exception as e: + _logger.exception('[ExcelParser] 解析失败: %s', e) + meta['error'] = f'解析失败:{e}' + return '', meta + + +def _detect_file_type(file_bytes: bytes, filename: str) -> str: + """根据文件内容和扩展名判断类型""" + # 1. 优先看扩展名 + lower_name = filename.lower() + if lower_name.endswith('.csv'): + return 'csv' + if lower_name.endswith('.xlsx'): + return 'xlsx' + if lower_name.endswith('.xls'): + return 'xls' + + # 2. 看魔数(文件头) + if file_bytes[:2] == b'PK': # ZIP (xlsx) + return 'xlsx' + if file_bytes[:2] == b'\xd0\xcf': # Old Excel (xls) + return 'xls' + # 尝试当作 CSV + try: + file_bytes.decode('utf-8-sig') + return 'csv' + except: + pass + return 'unknown' + + +def _parse_csv(bytes_data: bytes, max_rows: int, max_total: int, meta: dict) -> Tuple[str, dict]: + """解析 CSV 文件""" + import pandas as pd + + # 尝试多种编码 + encoding = _detect_encoding(bytes_data) + df = pd.read_csv(io.BytesIO(bytes_data), encoding=encoding, nrows=max_total + 1) + + return _df_to_markdown(df, 'CSV', max_rows, max_total, meta) + + +def _parse_excel(bytes_data: bytes, max_rows: int, max_total: int, meta: dict) -> Tuple[str, dict]: + """解析 Excel 文件(支持多 Sheet)""" + import pandas as pd + + xl = pd.ExcelFile(io.BytesIO(bytes_data)) + sheet_names = xl.sheet_names + + meta['sheets'] = sheet_names + + parts = [] + total_rows = 0 + + for sheet_name in sheet_names: + df = pd.read_excel(xl, sheet_name=sheet_name, nrows=max_total + 1 - total_rows) + + # 检查是否超出总行数 + if total_rows + len(df) > max_total: + df = df.head(max_total - total_rows) + meta['truncated'] = True + _logger.warning('[ExcelParser] 总行数超过 %s,已截断', max_total) + + sheet_md, sheet_meta = _df_to_markdown(df, sheet_name, max_rows, max_total, meta, is_single_sheet=(len(sheet_names) == 1)) + parts.append(sheet_md) + total_rows += len(df) + + if meta.get('truncated'): + break + + meta['total_rows'] = total_rows + return '\n\n'.join(parts), meta + + +def _df_to_markdown( + df, + sheet_name: str, + max_rows: int, + max_total: int, + meta: dict, + is_single_sheet: bool = False, +) -> Tuple[str, dict]: + """ + 将 DataFrame 转为 Markdown。 + 策略: + - 行数 <= max_rows:完整输出 + - 行数 > max_rows:输出统计摘要 + 前 max_rows 行 + """ + row_count = len(df) + col_count = len(df.columns) + + if not is_single_sheet: + parts = [f'## Sheet: {sheet_name}'] + else: + parts = [] + + parts.append(f'({row_count} 行 × {col_count} 列)\n') + + if row_count == 0: + parts.append('(空表)') + return '\n'.join(parts), meta + + # 统计摘要(数值列) + numeric_cols = df.select_dtypes(include=['number']).columns.tolist() + if numeric_cols: + parts.append('【数值列统计】') + for col in numeric_cols[:10]: # 最多10个数值列 + series = df[col].dropna() + if len(series) > 0: + parts.append(f' - {col}:最小={series.min():.2f},最大={series.max():.2f},平均={series.mean():.2f},合计={series.sum():.2f}') + parts.append('') + + # 分类列(唯一值数量) + cat_cols = [c for c in df.columns if c not in numeric_cols] + if cat_cols: + parts.append('【分类列概况】') + for col in cat_cols[:5]: # 最多5个分类列 + unique_count = df[col].nunique() + top_val = df[col].mode().iloc[0] if len(df[col].mode()) > 0 else '(无)' + parts.append(f' - {col}:共 {unique_count} 种值,最常见「{top_val}」') + parts.append('') + + # 数据内容 + if row_count <= max_rows: + parts.append('【完整数据】') + parts.append(df.to_markdown(index=False)) + else: + parts.append(f'【前 {max_rows} 行数据】(共 {row_count} 行,已截断)') + parts.append(df.head(max_rows).to_markdown(index=False)) + meta['truncated'] = True + + return '\n'.join(parts), meta + + +def _detect_encoding(bytes_data: bytes) -> str: + """检测 CSV 文件编码""" + for enc in ['utf-8-sig', 'utf-8', 'gbk', 'gb2312', 'iso-8859-1']: + try: + bytes_data.decode(enc) + return enc + except: + continue + return 'utf-8' + + +def validate_excel_file(file_bytes: bytes, filename: str) -> Dict[str, Any]: + """ + 验证 Excel 文件是否合法、大小是否超限。 + 返回:{valid, error, file_type, size_mb} + """ + size_mb = len(file_bytes) / 1024 / 1024 + + if size_mb > MAX_FILE_SIZE_MB: + return { + 'valid': False, + 'error': f'文件过大({size_mb:.1f}MB),上限 {MAX_FILE_SIZE_MB}MB', + 'file_type': '', + 'size_mb': round(size_mb, 1), + } + + file_type = _detect_file_type(file_bytes, filename) + if file_type == 'unknown': + return { + 'valid': False, + 'error': f'不支持的文件格式,仅支持 .xlsx / .xls / .csv', + 'file_type': file_type, + 'size_mb': round(size_mb, 1), + } + + return { + 'valid': True, + 'error': '', + 'file_type': file_type, + 'size_mb': round(size_mb, 1), + } diff --git a/yuthon_ai_agent/views/ai_import_record_views.xml b/yuthon_ai_agent/views/ai_import_record_views.xml new file mode 100644 index 000000000..fc2aa84f9 --- /dev/null +++ b/yuthon_ai_agent/views/ai_import_record_views.xml @@ -0,0 +1,207 @@ + + + + + ai.import.record.tree + ai.import.record + + + + + + + + + + + + + + + + ai.import.record.form + ai.import.record + +
+
+
+ +
+

+
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
+
+
+
+ + + ai.import.record.search + ai.import.record + + + + + + + + + + + + + + + + + + + + + + 附件导入记录 + ai.import.record + tree,form + + + + + + + + + ai.import.rule.tree + ai.import.rule + + + + + + + + + + + + + + + ai.import.rule.form + ai.import.rule + +
+
+
+ +
+

+
+ + + + + + + + + + + + + + + + + + + + + + + +
+
+
+
+ + + 导入规则配置 + ai.import.rule + tree,form + + + + +