# -*- coding: utf-8 -*- """ MinerU 统一文档解析模块 (v3.0+) 使用 MinerU v3.0+ 进行高质量多格式文档解析: - PDF: 表格识别率 95%+,支持 109 种语言 OCR - DOCX: 原生 Word 解析,速度提升数十倍,无幻觉 - XLSX: Excel 表格解析 - PPTX: PowerPoint 幻灯片解析 - 图片: 直接 OCR 识别 统一输出格式: - 保留标题层级(text_level) - 输出 Markdown + JSON 双格式 - 表格输出 HTML 格式 依赖: pip install "mineru[all]" mineru-models-download -s huggingface -m all """ import os # 配置 MinerU 设备模式(从 config.py 或环境变量读取) # 优先级:环境变量 > config.py 配置 > 默认值 cpu try: from config import MINERU_DEVICE_MODE if os.getenv('MINERU_DEVICE_MODE') is None: os.environ['MINERU_DEVICE_MODE'] = MINERU_DEVICE_MODE except ImportError: # config.py 不存在时回退到默认值 if os.getenv('MINERU_DEVICE_MODE') is None: os.environ['MINERU_DEVICE_MODE'] = 'cpu' import json import tempfile import shutil import subprocess import hashlib from pathlib import Path from typing import List, Dict, Optional, Any from dataclasses import dataclass, field from urllib.parse import urlparse import re import logging logger = logging.getLogger(__name__) # 文件大小限制 MAX_PDF_SIZE = 100 * 1024 * 1024 # 100MB # 支持的文件格式 SUPPORTED_FORMATS = { '.pdf': 'PDF 文档', '.docx': 'Word 文档', '.xlsx': 'Excel 表格', '.pptx': 'PowerPoint 幻灯片', '.png': 'PNG 图片', '.jpg': 'JPEG 图片', '.jpeg': 'JPEG 图片', '.bmp': 'BMP 图片', '.tiff': 'TIFF 图片', } def normalize_image_path(path: str) -> str: """ 规范化图片路径,处理各种边界情况 Args: path: 原始图片路径(可能包含 query 参数、相对路径等) Returns: 规范化后的文件名 """ # 去掉 query 参数 path = urlparse(path).path # 只保留文件名 return os.path.basename(path) def extract_images_from_markdown(content: str) -> List[Dict]: """ 从 Markdown/HTML 中提取图片引用 Args: content: Markdown 或 HTML 内容 Returns: [{"id": "abc.jpg", "order": 1}, ...] """ seen = {} # 用于去重保序 order = 0 # 匹配 Markdown 格式: ![alt](path) md_pattern = r'!\[([^\]]*)\]\(([^)]+)\)' for match in re.finditer(md_pattern, content): img_path = match.group(2) img_id = normalize_image_path(img_path) if img_id and img_id not in seen: order += 1 seen[img_id] = {"id": img_id, "order": order} # 匹配 HTML 格式: html_pattern = r']+src=["\']([^"\']+)["\']' for match in re.finditer(html_pattern, content): img_path = match.group(1) img_id = normalize_image_path(img_path) if img_id and img_id not in seen: order += 1 seen[img_id] = {"id": img_id, "order": order} return list(seen.values()) @dataclass class MinerUChunk: """MinerU 解析结果分块""" content: str # 文本内容 chunk_type: str # 类型: text, table, image, equation page_start: int = 1 # 起始页码 page_end: int = 1 # 结束页码 text_level: int = 0 # 标题级别 (0=body, 1=h1, 2=h2...) title: str = "" # 标题文本 section_path: str = "" # 章节路径 bbox: Optional[List[float]] = None # 边界框 [x0, y0, x1, y1] source_file: str = "" # 源文件名 table_html: Optional[str] = None # 表格 HTML(如果是表格) image_path: Optional[str] = None # 图片路径(独立图片) images: Optional[List[Dict]] = None # 关联图片列表: [{"id": "abc.jpg", "order": 1}] # 图片上下文(用于语义检索) context_before: str = "" # 图片前的文本上下文 context_after: str = "" # 图片后的文本上下文 # VLM 增强信息 vlm_description: str = "" # VLM 视觉描述(图片/图表) chart_markdown: str = "" # VLM 提取的图表数据表(Markdown 格式) # MinerU 结构化元数据 table_type: str = "" # 表格类型(cflow/table/text,来自 _v2_table_type) table_nest_level: str = "" # 表格嵌套层级(来自 _v2_table_nest_level) sub_type: str = "" # 图片/图表子类型(natural_image/table_image 等) def parse_with_mineru_online( file_path: str, api_token: str = None, api_url: str = None, model_version: str = None, timeout: int = None ) -> Dict[str, Any]: """ 使用 MinerU 在线 API 解析文档 当本地设备能力有限时,可使用在线 API 作为备选方案。 需要在 https://mineru.net/apiManage/token 申请 API Token。 API 流程: 1. 申请上传链接 → /api/v4/file-urls/batch 2. PUT 上传文件 3. 系统自动提交解析任务 4. 轮询查询结果 → /api/v4/extract-results/batch/{batch_id} 5. 下载 zip 包并解析 Args: file_path: 文档文件路径 api_token: API Token(默认从 config 读取) api_url: API 地址 model_version: 模型版本 (vlm / pipeline / MinerU-HTML),默认从 config 读取 timeout: 轮询超时(秒),默认从 config 读取 Returns: 解析结果(与 parse_with_mineru 格式相同) """ import requests import time import zipfile import io from config import MINERU_API_TOKEN, MINERU_API_URL, MINERU_MODEL_VERSION, MINERU_ONLINE_TIMEOUT token = api_token or MINERU_API_TOKEN url = api_url or MINERU_API_URL model_version = model_version or MINERU_MODEL_VERSION timeout = timeout or MINERU_ONLINE_TIMEOUT if not token: raise RuntimeError("MinerU 在线 API Token 未配置,请在 config.py 中设置 MINERU_API_TOKEN") file_path = Path(file_path) if not file_path.exists(): raise FileNotFoundError(f"文件不存在: {file_path}") # 检查文件大小 file_size = file_path.stat().st_size if file_size > MAX_PDF_SIZE: raise ValueError(f"文件过大: {file_size / 1024 / 1024:.1f}MB,最大允许 {MAX_PDF_SIZE / 1024 / 1024:.0f}MB") logger.info(f"使用 MinerU 在线 API 解析: {file_path.name}") # 读取文件内容 with open(file_path, 'rb') as f: file_content = f.read() headers = { "Content-Type": "application/json", "Authorization": f"Bearer {token}" } try: # 1. 申请上传链接 upload_url = url.replace("/extract/task", "/file-urls/batch") upload_data = { "files": [{"name": file_path.name, "data_id": "upload"}], "model_version": model_version } resp = requests.post(upload_url, json=upload_data, headers=headers, timeout=30) resp.raise_for_status() upload_result = resp.json() if upload_result.get("code") != 0: raise RuntimeError(f"获取上传凭证失败: {upload_result}") data = upload_result.get("data", {}) batch_id = data.get("batch_id") file_urls = data.get("file_urls", []) if not file_urls or not batch_id: raise RuntimeError(f"未获取到上传 URL 或 batch_id: {upload_result}") logger.info(f"获取上传链接成功, batch_id: {batch_id}") # 2. PUT 上传文件 put_url = file_urls[0] put_resp = requests.put(put_url, data=file_content, timeout=timeout) put_resp.raise_for_status() logger.info(f"文件上传成功: {file_path.name}") # 3. 轮询等待结果(系统自动提交任务) # 查询接口: /api/v4/extract-results/batch/{batch_id} result_url = url.replace("/extract/task", f"/extract-results/batch/{batch_id}") max_wait = timeout poll_interval = 5 waited = 0 while waited < max_wait: time.sleep(poll_interval) waited += poll_interval result_resp = requests.get(result_url, headers=headers, timeout=30) result_resp.raise_for_status() result = result_resp.json() # 检查 API 层面的错误码,快速失败而非静默等到超时 api_code = result.get("code") if api_code and api_code != 0: api_msg = result.get("msg", "未知错误") raise RuntimeError(f"MinerU API 错误 (code={api_code}): {api_msg}") extract_results = result.get("data", {}).get("extract_result", []) if not extract_results: logger.debug(f"等待解析结果... ({waited}s/{max_wait}s)") continue # 取第一个文件的结果 file_result = extract_results[0] state = file_result.get("state", "") if state == "done": zip_url = file_result.get("full_zip_url") if not zip_url: raise RuntimeError(f"解析完成但未获取到结果 URL: {file_result}") logger.info(f"MinerU 在线解析完成,下载结果...") # 4. 下载 zip 包 zip_resp = requests.get(zip_url, timeout=120) zip_resp.raise_for_status() # 5. 解析 zip 包内容 result = _parse_mineru_online_zip(zip_resp.content, file_path) # 保存 zip 内容,供后续提取图片使用 result['_zip_content'] = zip_resp.content return result elif state == "failed": error = file_result.get("err_msg", "未知错误") raise RuntimeError(f"MinerU 在线解析失败: {error}") else: # waiting-file / pending / running / converting progress = file_result.get("extract_progress", {}) if progress: extracted = progress.get("extracted_pages", 0) total = progress.get("total_pages", 0) logger.info(f"解析进度: {extracted}/{total} 页 ({waited}s/{max_wait}s, model={model_version})") else: logger.debug(f"状态: {state}, 等待中... ({waited}s/{max_wait}s)") raise RuntimeError( f"MinerU 在线解析超时 ({max_wait}s)" f",当前 model_version={model_version}" f",可尝试: 1) 设置 MINERU_MODEL_VERSION=pipeline 加速" f" 2) 增大 MINERU_ONLINE_TIMEOUT" ) except Exception as e: raise RuntimeError(f"MinerU 在线 API 调用失败: {e}") def _parse_v2_content_list(v2_data: list) -> list: """ 将 MinerU v2 嵌套格式转换为 v1 兼容的扁平 content_list v2 格式: [[page0_items], [page1_items], ...] v1 格式: [item, item, ...] 扁平列表 转换规则: - paragraph → text(拼接 paragraph_content,提取 style 信息) - title → text(从 title_content 提取文本,level 信息) - table → table(保留 html,提取 table_caption/table_footnote 列表格式) - image → image(提取 image_source.path、VLM 视觉描述 content.content) - chart → chart(提取 image_source.path、VLM 数据表 content.content Markdown) - list → text(将 list_items 拼接为段落,保留 list_type) - equation → equation - page_header / page_footer / page_number → 过滤掉 支持的额外字段(_v2_ 前缀): - _v2_styles: 样式列表(如 ['bold']) - _v2_table_type / _v2_table_nest_level / _v2_table_footnote: 表格元数据 - _v2_vlm_description: VLM 生成的图片视觉描述 - _v2_chart_markdown: VLM 从图表中提取的 Markdown 数据表 - _v2_list_type: 列表类型(text_list 等) Args: v2_data: v2 格式的嵌套列表 Returns: v1 兼容的扁平列表 """ flat_list: List[Dict] = [] for page_idx, page_items in enumerate(v2_data): if not isinstance(page_items, list): continue for item in page_items: if not isinstance(item, dict): continue v2_type: str = item.get('type', '') content = item.get('content', {}) # 过滤噪音类型(页眉、页脚、页码、目录索引) if v2_type in ('page_header', 'page_footer', 'page_number', 'index'): continue if v2_type == 'paragraph': # 拼接段落文本,收集样式信息 text_parts: List[str] = [] has_bold = False para_content = content.get('paragraph_content', []) if isinstance(content, dict) else [] for part in para_content: if not isinstance(part, dict): continue part_text = part.get('content', '') text_parts.append(part_text) if 'bold' in part.get('style', []): has_bold = True full_text = ''.join(text_parts).strip() if not full_text: continue flat_item: Dict[str, Any] = { 'type': 'text', 'text': full_text, 'content': full_text, 'page_idx': page_idx, 'bbox': item.get('bbox', []), 'text_level': 0, '_v2_styles': ['bold'] if has_bold else [], } flat_list.append(flat_item) elif v2_type == 'table': # table_caption 在 V2 中是列表格式: [{"type": "text", "content": "..."}] table_caption_list = content.get('table_caption', []) if isinstance(content, dict) else [] table_caption = ''.join( p.get('content', '') for p in table_caption_list if isinstance(p, dict) ).strip() if isinstance(table_caption_list, list) else str(table_caption_list) # 兜底: 如果 caption 列表为空,尝试旧的 caption 字符串字段 if not table_caption: table_caption = content.get('caption', '') if isinstance(content, dict) else '' # table_footnote table_footnote_list = content.get('table_footnote', []) if isinstance(content, dict) else [] table_footnote = ''.join( p.get('content', '') for p in table_footnote_list if isinstance(p, dict) ).strip() if isinstance(table_footnote_list, list) else '' # image_source (V2 新格式) 兜底 img_path img_src = content.get('image_source', {}) if isinstance(content, dict) else {} img_path = img_src.get('path', '') if isinstance(img_src, dict) else '' if not img_path: img_path = content.get('img_path', '') if isinstance(content, dict) else '' flat_item = { 'type': 'table', 'page_idx': page_idx, 'bbox': item.get('bbox', []), 'html': content.get('html', '') if isinstance(content, dict) else '', 'table_body': content.get('html', '') if isinstance(content, dict) else '', 'caption': table_caption, 'img_path': img_path, '_v2_table_type': content.get('table_type', '') if isinstance(content, dict) else '', '_v2_table_nest_level': content.get('table_nest_level', '') if isinstance(content, dict) else '', '_v2_table_footnote': table_footnote, } flat_list.append(flat_item) elif v2_type == 'title': # v2 title 项: level 在 content.level,文本在 title_content(非 paragraph_content) vlm_level = content.get('level', 0) if isinstance(content, dict) else 0 text_parts = [] # PDF V2 使用 title_content,DOCX 理论上不应出现 title 类型 title_content = content.get('title_content', []) if isinstance(content, dict) else [] # 兜底: 如果 title_content 为空,尝试 paragraph_content if not title_content: title_content = content.get('paragraph_content', []) if isinstance(content, dict) else [] for part in title_content: if isinstance(part, dict): text_parts.append(part.get('content', '')) full_text = ''.join(text_parts).strip() if full_text: # PDF V2: heading_rules 优先(模式匹配对编号标题可靠) # VLM level 仅作兜底(VLM 常给所有标题 level=1,不可靠) level = 0 try: from parsers.heading_rules import get_heading_engine engine = get_heading_engine() detected_level, rule_name = engine.detect(full_text, style=['bold']) if detected_level > 0: level = detected_level except Exception: pass if level == 0 and vlm_level > 0: level = vlm_level flat_item = { 'type': 'text', 'text': full_text, 'content': full_text, 'page_idx': page_idx, 'bbox': item.get('bbox', []), 'text_level': level, '_v2_styles': ['bold'], } flat_list.append(flat_item) elif v2_type in ('image', 'chart'): # image_source (V2 新格式) 兜底 img_path img_src = content.get('image_source', {}) if isinstance(content, dict) else {} img_path = img_src.get('path', '') if isinstance(img_src, dict) else '' if not img_path: img_path = content.get('img_path', '') if isinstance(content, dict) else '' # caption 在 V2 中可能是列表格式 caption_raw = content.get('image_caption', content.get('caption', '')) if isinstance(content, dict) else '' if isinstance(caption_raw, list): caption = ''.join(p.get('content', '') for p in caption_raw if isinstance(p, dict)).strip() else: caption = str(caption_raw) # VLM 视觉描述 (image 和 chart 项的 content.content 字段) vlm_description = '' if isinstance(content, dict): desc = content.get('content', '') if isinstance(desc, str) and desc and len(desc) > 10: # image 直接使用;chart 需排除 markdown 表格(表格走 chart_markdown) if v2_type == 'image': vlm_description = desc elif v2_type == 'chart' and '|' not in desc: vlm_description = desc # chart 的 VLM 数据表 (content.content 字段,Markdown 表格) chart_markdown = '' if v2_type == 'chart' and isinstance(content, dict): md = content.get('content', '') if isinstance(md, str) and md: if '|' in md: chart_markdown = md # 即使没有 '|',如果有结构化数据特征也保留 elif len(md) > 50 and any(kw in md for kw in ('数据', '合计', '总计', '年份', '单位')): chart_markdown = md # chart caption chart_caption_list = content.get('chart_caption', []) if isinstance(chart_caption_list, list) and chart_caption_list: caption = ''.join( p.get('content', '') for p in chart_caption_list if isinstance(p, dict) ).strip() or caption # sub_type (natural_image / table_image 等) sub_type = item.get('sub_type', '') # 封面 logo 过滤:第一页无 caption 无 VLM 描述的图片通常是封面装饰 if (v2_type == 'image' and page_idx == 0 and not caption and not vlm_description and not img_path): continue flat_item = { 'type': v2_type, 'page_idx': page_idx, 'bbox': item.get('bbox', []), 'img_path': img_path, 'image_path': img_path, 'caption': caption, 'sub_type': sub_type, '_v2_vlm_description': vlm_description, '_v2_chart_markdown': chart_markdown, } flat_list.append(flat_item) elif v2_type == 'list': # 结构化列表:将列表项拼接为段落文本 list_items = content.get('list_items', []) if isinstance(content, dict) else [] item_texts = [] for li in list_items: if not isinstance(li, dict): continue item_content = li.get('item_content', []) text = ''.join( p.get('content', '') for p in item_content if isinstance(p, dict) ).strip() if text: item_texts.append(text) if item_texts: list_type = content.get('list_type', 'text_list') if isinstance(content, dict) else 'text_list' full_text = '\n'.join(item_texts) flat_item = { 'type': 'text', 'text': full_text, 'content': full_text, 'page_idx': page_idx, 'bbox': item.get('bbox', []), 'text_level': 0, '_v2_styles': [], '_v2_list_type': list_type, } flat_list.append(flat_item) elif v2_type == 'equation': flat_item = { 'type': 'equation', 'page_idx': page_idx, 'bbox': item.get('bbox', []), 'content': content.get('latex', '') if isinstance(content, dict) else '', 'text': content.get('latex', '') if isinstance(content, dict) else '', 'latex': content.get('latex', '') if isinstance(content, dict) else '', 'img_path': content.get('img_path', '') if isinstance(content, dict) else '', } flat_list.append(flat_item) # ── TOC 残留过滤 ── # 目录条目可能以 list / title / paragraph 类型混入,用行尾页码模式检测 # 模式1: 连续点号/省略号+页码(如 "1 综述..........1"、"2 三峡工程…4") # 模式2: 短行+空格+页码数字(如 "2.2 防洪 8"、"6.2 水位 28") import re _toc_dots = re.compile(r'(\.{2,}|…+|⋯+)\s*\d+\s*$') _toc_space_num = re.compile(r'\s{2,}\d{1,3}\s*$') # 2+空格+1-3位数字 _toc_filtered = 0 filtered_list = [] for fi in flat_list: text = fi.get('text', '') or fi.get('content', '') # 只检查 text 类型(title/list/paragraph 产出),table/image/chart 不动 if fi.get('type') == 'text' and text: lines = [l.strip() for l in text.split('\n') if l.strip()] if lines: toc_hits = 0 for l in lines: if _toc_dots.search(l): toc_hits += 1 elif _toc_space_num.search(l) and len(l) < 40: # 短行+空格+页码:典型的目录格式 toc_hits += 1 # TOC 判定逻辑: # 1) 匹配率 > 50% → 确定是目录 # 2) 匹配率 > 25% 且平均行长 < 25字 → 目录(短行+页码是强信号) avg_line_len = sum(len(l) for l in lines) / len(lines) if lines else 0 match_ratio = toc_hits / len(lines) if lines else 0 is_toc = (match_ratio > 0.5) or (match_ratio > 0.25 and avg_line_len < 25) if is_toc: _toc_filtered += 1 logger.debug(f"TOC 过滤命中({toc_hits}/{len(lines)}, avg={avg_line_len:.0f}): {text[:60]}...") continue elif toc_hits > 0: logger.debug(f"TOC 部分匹配({toc_hits}/{len(lines)}): {text[:60]}...") filtered_list.append(fi) if _toc_filtered > 0: logger.info(f"TOC 过滤第一轮: 移除 {_toc_filtered} 个目录块") flat_list = filtered_list # 第二轮:清理孤立的 TOC 标题(子条目被过滤后残留的父标题) _toc_orphan = 0 final_list = [] for fi in flat_list: text = (fi.get('text', '') or fi.get('content', '')).strip() page = fi.get('page_idx', 0) if fi.get('type') == 'text' and text and page <= 5: # 孤立 "目录" 标题 if text == '目录': _toc_orphan += 1 continue # 短标题 + 尾部页码数字(如 "6 长江中下游河道状况 25") if (len(text) < 40 and not text.endswith('。') and not text.endswith(';') and re.search(r'\s+\d{1,3}\s*$', text)): _toc_orphan += 1 continue final_list.append(fi) if _toc_orphan > 0: logger.info(f"TOC 过滤第二轮: 移除 {_toc_orphan} 个孤立目录标题") flat_list = final_list # ── 单字符标题残留过滤 ── # VLM 可能部分识别目录标题(如 "目录" → "录"),单字符标题几乎不会是有效章节 _single_char = 0 _sc_list = [] for fi in flat_list: text = (fi.get('text', '') or fi.get('content', '')).strip() if fi.get('type') == 'text' and text and len(text) <= 1 and fi.get('text_level', 0) > 0: _single_char += 1 logger.debug(f"单字符标题过滤: '{text}' pg={fi.get('page_idx', 0)}") continue _sc_list.append(fi) if _single_char > 0: logger.info(f"单字符标题过滤: 移除 {_single_char} 个残留") flat_list = _sc_list # ── 封面重复标题去重 ── # pg=0(封面)和 pg=1(扉页)常有完全相同的标题(如 "三峡工程公报"、"2022"),保留较后的 _cover_seen = {} # text -> page_idx _cover_dup = 0 dedup_list = [] for fi in flat_list: text = (fi.get('text', '') or fi.get('content', '')).strip() page = fi.get('page_idx', 0) if fi.get('type') == 'text' and text and page <= 1 and fi.get('text_level', 0) > 0: if text in _cover_seen: _cover_dup += 1 logger.debug(f"封面重复标题过滤: '{text}' pg={page} (首次 pg={_cover_seen[text]})") continue _cover_seen[text] = page dedup_list.append(fi) if _cover_dup > 0: logger.info(f"封面去重: 移除 {_cover_dup} 个重复封面标题") flat_list = dedup_list logger.info(f"v2 格式转换: {len(v2_data)} 页 → {len(flat_list)} 项(已过滤噪音类型)") return flat_list def _parse_mineru_online_zip(zip_content: bytes, file_path: Path) -> Dict[str, Any]: """ 解析 MinerU 在线 API 返回的 zip 包 zip 包结构: - full.md - Markdown 解析结果 - *_content_list.json - 内容列表(v1 扁平格式) - *_content_list_v2.json - 内容列表(v2 嵌套格式,含 style 信息) - images/ - 图片目录 Args: zip_content: zip 文件二进制内容 file_path: 源文件路径 Returns: 解析结果(与 _parse_mineru_output 格式相同) """ import zipfile import io # 读取格式偏好配置 try: from config import MINERU_PREFER_V2 except ImportError: MINERU_PREFER_V2 = True # 默认优先 v2 with zipfile.ZipFile(io.BytesIO(zip_content), 'r') as zf: # 列出所有文件 file_list = zf.namelist() logger.debug(f"zip 包内容: {file_list}") # 查找 content_list 文件(v2 含 style 信息,优先使用) content_list_path = None is_v2 = False if MINERU_PREFER_V2: # 优先使用 v2 格式(含 style 信息) for f in file_list: if f.endswith('_content_list_v2.json'): content_list_path = f is_v2 = True break if not content_list_path: for f in file_list: if f.endswith('_content_list.json') and not f.endswith('_v2.json'): content_list_path = f break else: # 优先使用 v1 格式 for f in file_list: if f.endswith('_content_list.json') and not f.endswith('_v2.json'): content_list_path = f break if not content_list_path: for f in file_list: if f.endswith('_content_list_v2.json'): content_list_path = f is_v2 = True break # 查找 markdown 文件 md_path = None for f in file_list: if f.endswith('.md'): md_path = f break # 读取 content_list content_list = [] if content_list_path: with zf.open(content_list_path) as f: content_list = json.load(f) # v2 格式需要转换为 v1 兼容的扁平列表 if is_v2 and isinstance(content_list, list) and content_list and isinstance(content_list[0], list): content_list = _parse_v2_content_list(content_list) logger.info(f"v2 格式已转换为扁平列表,共 {len(content_list)} 项") logger.info(f"读取 content_list: {len(content_list)} 项, 格式={'v2' if is_v2 else 'v1'}, 来源: {content_list_path}") # 读取 markdown markdown_content = "" if md_path: with zf.open(md_path) as f: markdown_content = f.read().decode('utf-8') # 提取图片路径 images = [] for f in file_list: # 检查是否是 images 目录下的文件(支持 images/xxx 或 /images/xxx 格式) if ('images/' in f or f.startswith('images/')) and not f.endswith('/'): images.append(f) # 如果有 content_list,使用 _parse_mineru_online_result 解析 if content_list: result = { "data": { "content_list": content_list, "markdown": markdown_content } } parsed_result = _parse_mineru_online_result(result, file_path) # 添加 zip 内图片列表,供后续提取使用 parsed_result['_zip_images'] = images return parsed_result # 否则只返回 markdown return { 'markdown': markdown_content, 'chunks': [], 'tables': [], 'images': images, 'content_list': [], '_zip_images': images } def _parse_mineru_online_result(result: Dict, file_path: Path) -> Dict[str, Any]: """ 解析 MinerU 在线 API 返回结果 保持与本地 _parse_mineru_output 完全一致的输出格式, 确保后续处理流程不受影响。 """ data = result.get("data", {}) markdown_content = data.get("markdown", "") or data.get("full_markdown", "") content_list = data.get("content_list", []) logger.info(f"解析在线结果: content_list 项数={len(content_list)}, markdown 长度={len(markdown_content)}") chunks = [] tables = [] images = [] section_stack = [] # 追踪章节层级 markdown_parts = [] # 第一遍扫描:收集文本内容(用于构建图片上下文) text_items = [] for idx, item in enumerate(content_list): if item.get("type") == "text": text = item.get("content", "") or item.get("text", "") if text: text_items.append((idx, text.strip(), item.get("page_idx", 0))) def get_context_for_image(image_idx: int, page_idx: int, window: int = 3) -> tuple: """获取图片前后的文本上下文""" context_before = [] context_after = [] for item_idx, text, item_page in text_items: if item_idx < image_idx and item_page >= page_idx - 1: context_before.append(text) elif item_idx > image_idx and item_page <= page_idx + 1: context_after.append(text) return " ".join(context_before[-window:]), " ".join(context_after[:window]) # 解析 content_list for idx, item in enumerate(content_list): item_type = item.get("type", "text") page_idx = item.get("page_idx", 0) bbox = item.get("bbox", []) text_level = item.get("text_level", 0) if item_type == "text": text = item.get("content", "") or item.get("text", "") v2_styles = item.get("_v2_styles", []) # 启发式标题识别 if text_level == 0: from parsers.heading_rules import get_heading_engine engine = get_heading_engine() detected_level, _ = engine.detect(text, style=v2_styles) text_level = detected_level title = "" if text_level > 0: title = text.strip() while section_stack and section_stack[-1][0] >= text_level: section_stack.pop() section_stack.append((text_level, title)) section_path = " > ".join([s[1] for s in section_stack]) if text_level > 0: md_line = f"{'#' * text_level} {text}" else: md_line = text markdown_parts.append(md_line) chunk = MinerUChunk( content=text, chunk_type="heading" if text_level > 0 else "text", page_start=page_idx + 1, page_end=page_idx + 1, text_level=text_level, title=title, section_path=section_path, bbox=bbox, source_file=file_path.name ) chunks.append(chunk) elif item_type == "table": table_body = item.get("html", "") or item.get("table_body", "") table_caption = item.get("caption", "") or item.get("table_caption", "") table_footnote = item.get("_v2_table_footnote", "") img_path = item.get("img_path", "") or item.get("image_path", "") section_path = " > ".join([s[1] for s in section_stack]) markdown_parts.append(f"\n| 表格 |") if table_body: md_table = html_table_to_markdown(table_body) markdown_parts.append(md_table) else: md_table = "" # 从原始 HTML 提取嵌入图片(md_table 经 get_text 转换后已丢失 标签) table_images = extract_images_from_markdown(table_body) if table_body else [] # 表格内容增强:caption + footnote table_content = table_caption or "表格" if table_footnote: table_content = f"{table_content}\n[脚注] {table_footnote}" chunk = MinerUChunk( content=table_content, chunk_type="table", page_start=page_idx + 1, page_end=page_idx + 1, title=table_caption or "表格", section_path=section_path, bbox=bbox, source_file=file_path.name, table_html=table_body, image_path=img_path, images=table_images if table_images else None, table_type=item.get("_v2_table_type", ""), table_nest_level=item.get("_v2_table_nest_level", ""), ) chunks.append(chunk) if table_body: tables.append(table_body) if img_path: images.append(img_path) elif item_type in ("image", "chart"): img_path = item.get("img_path", "") or item.get("image_path", "") caption = item.get("caption", "") vlm_desc = item.get("_v2_vlm_description", "") chart_md = item.get("_v2_chart_markdown", "") sub_type = item.get("sub_type", "") section_path = " > ".join([s[1] for s in section_stack]) markdown_parts.append(f"\n![{caption}]({img_path})") # chart 的 VLM 数据表也写入 markdown 输出 if chart_md: markdown_parts.append(chart_md) chunk_type = "chart" if item_type == "chart" else "image" context_before, context_after = get_context_for_image(idx, page_idx) # 图片内容增强:caption + VLM 描述 content_text = caption or ("图表" if item_type == "chart" else "图片") if vlm_desc: content_text = f"{content_text}\n[视觉描述] {vlm_desc}" if chart_md: content_text = f"{content_text}\n[数据表]\n{chart_md}" chunk = MinerUChunk( content=content_text, chunk_type=chunk_type, page_start=page_idx + 1, page_end=page_idx + 1, title=caption or ("图表" if item_type == "chart" else "图片"), section_path=section_path, bbox=bbox, source_file=file_path.name, image_path=img_path, context_before=context_before, context_after=context_after, vlm_description=vlm_desc, chart_markdown=chart_md, table_html=chart_md if chart_md else None, # chart 数据表作为表格存储 sub_type=sub_type, ) chunks.append(chunk) if img_path: images.append(img_path) # chart 数据表也加入 tables 列表,便于表格检索 if chart_md: tables.append(chart_md) elif item_type == "equation": # 处理公式类型 equation_content = item.get("content", "") or item.get("latex", "") or item.get("text", "") img_path = item.get("img_path", "") or item.get("image_path", "") section_path = " > ".join([s[1] for s in section_stack]) if equation_content: markdown_parts.append(f"\n$$ {equation_content} $$") chunk = MinerUChunk( content=equation_content or "公式", chunk_type="equation", page_start=page_idx + 1, page_end=page_idx + 1, title="公式", section_path=section_path, bbox=bbox, source_file=file_path.name, image_path=img_path ) chunks.append(chunk) if img_path: images.append(img_path) # 如果没有解析到内容,使用 Markdown if not chunks and markdown_content: markdown_parts = [markdown_content] # 后处理:与本地解析完全一致 try: from config import MIN_CHUNK_SIZE, MAX_CHUNK_SIZE min_merge = MIN_CHUNK_SIZE // 2 max_size = MAX_CHUNK_SIZE except ImportError: min_merge = 100 max_size = 1200 # 表单类型二次校正(在 _post_process_chunks 之前,因为 table 不参与合并) _reclassify_text_chunks(chunks) chunks = _post_process_chunks(chunks, min_merge_size=min_merge, max_chunk_size=max_size) # 验证分类标题是否被规则引擎正确识别(安全网,仅告警不修改) _validate_category_section_paths(chunks) return { 'markdown': "\n".join(markdown_parts) if markdown_parts else markdown_content, 'chunks': chunks, 'tables': tables, 'images': images, 'content_list': content_list } def parse_with_mineru( file_path: str, output_dir: Optional[str] = None, lang: str = "ch", enable_table: bool = True, enable_formula: bool = True, backend: str = None, start_page: int = 0, end_page: int = 99999 ) -> Dict[str, Any]: """ 使用 MinerU 解析文档(支持 PDF、DOCX、XLSX、PPTX、图片) Args: file_path: 文档文件路径 output_dir: 输出目录,默认使用临时目录 lang: 语言代码 (ch, en, etc.) enable_table: 启用表格识别 enable_formula: 启用公式识别 backend: 解析后端 - "pipeline": 通用模式(推荐) - "vlm-auto-engine": 高精度模式 - "hybrid-auto-engine": 新一代高精度方案 start_page: 起始页码(0-indexed,仅 PDF 有效) end_page: 结束页码(仅 PDF 有效) Returns: { 'markdown': str, # Markdown 内容 'chunks': List[MinerUChunk], # 结构化分块 'tables': List[str], # 表格列表 'images': List[str], # 图片列表 'content_list': List[Dict] # 原始 content_list } """ file_path = Path(file_path) if not file_path.exists(): raise FileNotFoundError(f"文件不存在: {file_path}") # 从 config 读取默认 backend if backend is None: try: from config import MINERU_LOCAL_BACKEND backend = MINERU_LOCAL_BACKEND except ImportError: backend = 'pipeline' # 检查文件大小 file_size = file_path.stat().st_size if file_size > MAX_PDF_SIZE: raise ValueError(f"文件过大: {file_size / 1024 / 1024:.1f}MB,最大允许 {MAX_PDF_SIZE / 1024 / 1024:.0f}MB") # 检查文件格式 suffix = file_path.suffix.lower() if suffix not in SUPPORTED_FORMATS: raise ValueError( f"不支持的文件格式: {suffix}。" f"支持格式: {', '.join(SUPPORTED_FORMATS.keys())}" ) logger.info(f"使用 MinerU 解析 {SUPPORTED_FORMATS.get(suffix, '文档')}: {file_path.name}") # 创建输出目录 if output_dir is None: output_dir = tempfile.mkdtemp(prefix="mineru_") cleanup_output = True else: output_dir = Path(output_dir) output_dir.mkdir(parents=True, exist_ok=True) cleanup_output = False # 构建 mineru 命令 # 使用虚拟环境中的 mineru import sys venv_dir = Path(sys.executable).parent mineru_exe = venv_dir / "mineru.exe" if not mineru_exe.exists(): mineru_exe = venv_dir / "mineru" if not mineru_exe.exists(): mineru_exe = "mineru" # 回退到系统 PATH # 参数白名单校验,防止注入非法参数 ALLOWED_BACKENDS = {'auto', 'pipeline', 'vlm', 'vlm-sglang', 'vlm-auto-engine', 'hybrid-auto-engine', 'ocr'} ALLOWED_LANGS = {'ch', 'en', 'ch_lite', 'en_lite', 'formula', 'table'} if backend not in ALLOWED_BACKENDS: backend = 'pipeline' if lang not in ALLOWED_LANGS: lang = 'ch' cmd = [ str(mineru_exe), "-p", str(file_path), "-o", str(output_dir), "-m", "auto", "-b", backend, "-l", lang, "-s", str(start_page), "-e", str(end_page) if end_page < 99999 else str(99999), "-f", str(enable_formula).lower(), "-t", str(enable_table).lower() ] try: # 执行 MinerU 命令 result = subprocess.run( cmd, capture_output=True, # 使用系统默认编码,避免 UTF-8 解码错误 encoding=None, errors='replace', timeout=600 # 10分钟超时 ) if result.returncode != 0: # 安全解码 stderr if result.stderr: stderr_output = result.stderr.decode('utf-8', errors='replace') if isinstance(result.stderr, bytes) else str(result.stderr) else: stderr_output = "" logger.error(f"MinerU 解析失败: {stderr_output}") raise RuntimeError(f"MinerU 解析失败: {stderr_output}") # 解析输出结果 return _parse_mineru_output(file_path, output_dir) except subprocess.TimeoutExpired: raise RuntimeError("MinerU 解析超时") except FileNotFoundError: raise RuntimeError( "MinerU 未安装或不在 PATH 中,请运行: pip install \"mineru[all]\"" ) except Exception as e: logger.error(f"MinerU 解析失败: {e}") raise finally: # 清理临时目录 if cleanup_output and os.path.exists(output_dir): shutil.rmtree(output_dir, ignore_errors=True) def _detect_heading_level(text: str) -> int: """ 启发式标题识别(规则引擎版) 当 MinerU 解析 DOCX 等 Office 格式时不提供 text_level 时使用。 规则按优先级从高到低匹配,第一个命中即返回。 规则定义见 parsers/heading_rules.py,可通过 config.py 覆盖。 Args: text: 文本内容 Returns: 标题级别 (0=正文, 1=h1, 2=h2, 3=h3) """ from parsers.heading_rules import get_heading_engine engine = get_heading_engine() level, _ = engine.detect(text) return level def _reclassify_text_chunks(chunks: List[MinerUChunk]) -> None: """ 二次校正:检测被标记为 text 但实际是表格/表单的 chunk MinerU 解析 Word 文档时,某些带下划线填空项的表单 被标记为 text 类型,需要根据内容特征修正为 table。 就地修改 chunks 列表中的 chunk_type 字段。 检测依据(基于实测数据设计): - 连续下划线 ___ (3个以上) —— Word 表单填空项 - 冒号后跟下划线 如 "日期:____" —— 键值对式表单 需 >= min_indicators 个指标同时命中才校正,避免误判。 """ try: from config import FORM_RECLASSIFY_ENABLED if not FORM_RECLASSIFY_ENABLED: return except ImportError: pass try: from config import FORM_RECLASSIFY_MIN_INDICATORS min_indicators = FORM_RECLASSIFY_MIN_INDICATORS except ImportError: min_indicators = 2 # 表单特征指标(编译一次,避免循环内重复编译) # 注意:MinerU 输出的下划线是 Markdown 转义格式 \_\_\_, # 需要同时匹配纯下划线 ___ 和转义下划线 \_\_\_ _FORM_INDICATORS = [ re.compile(r'(?:\\_|_){3,}'), # 连续下划线(3个以上,含转义格式) re.compile(r'[::]\s*(?:\\_|_){2,}'), # 冒号后跟下划线(含转义格式) ] reclassified = 0 for chunk in chunks: if chunk.chunk_type != 'text': continue text = (chunk.content or '').strip() if not text: continue indicator_count = sum(1 for p in _FORM_INDICATORS if p.search(text)) if indicator_count >= min_indicators: chunk.chunk_type = 'table' reclassified += 1 logger.info(f"表单检测: text -> table, content='{text[:80]}'") if reclassified > 0: logger.info(f"表单类型二次校正: 共 {reclassified} 个 text -> table") def _validate_category_section_paths(chunks: List[MinerUChunk]) -> None: """ 验证分类标题是否被规则引擎正确识别(安全网) 当规则引擎正确识别分类标题后,section_stack 自然会更新, 不再需要后处理修改 section_path。此函数仅做验证和告警, 便于发现规则引擎的遗漏。 支持的模式:A1类:、B2类:、C1类:等(含可选 ** 粗体标记)。 """ cat_pattern = re.compile(r'^\*{0,2}[A-Z]\d+[类類]\*{0,2}[::]') missed_count = 0 for chunk in chunks: if chunk.chunk_type == 'text': text = (chunk.content or '').strip() if cat_pattern.match(text) and chunk.text_level == 0: missed_count += 1 logger.warning( f"分类标题未被识别为标题: '{text[:50]}', " f"section_path='{chunk.section_path}'" ) if missed_count > 0: logger.warning( f"发现 {missed_count} 个分类标题未被规则引擎识别," f"请检查 heading_rules 配置" ) def _parse_mineru_output(file_path: Path, output_dir) -> Dict[str, Any]: """ 解析 MinerU 输出结果 Args: file_path: 源文件路径 output_dir: 输出目录 Returns: 解析结果字典 """ chunks = [] tables = [] images = [] section_stack = [] # 追踪章节层级 markdown_parts = [] # 确保 output_dir 是 Path 对象 output_dir = Path(output_dir) # 查找输出文件 # 不同格式的输出目录不同: # - PDF: output/文件名/auto/ # - DOCX/XLSX/PPTX: output/文件名/office/ doc_name = file_path.stem auto_dir = output_dir / doc_name / "auto" office_dir = output_dir / doc_name / "office" # 选择正确的输出目录 if auto_dir.exists(): output_subdir = auto_dir elif office_dir.exists(): output_subdir = office_dir else: raise RuntimeError(f"MinerU 输出目录不存在: {auto_dir} 或 {office_dir}") # 读取 content_list(v2 含 style 信息,优先使用) try: from config import MINERU_PREFER_V2 except ImportError: MINERU_PREFER_V2 = True v1_path = output_subdir / f"{doc_name}_content_list.json" v2_path = output_subdir / f"{doc_name}_content_list_v2.json" content_list_path = None is_v2 = False if MINERU_PREFER_V2: # 优先使用 v2 格式 if v2_path.exists(): content_list_path = v2_path is_v2 = True elif v1_path.exists(): content_list_path = v1_path else: # 优先使用 v1 格式 if v1_path.exists(): content_list_path = v1_path elif v2_path.exists(): content_list_path = v2_path is_v2 = True content_list = [] if content_list_path and content_list_path.exists(): with open(content_list_path, 'r', encoding='utf-8') as f: content_list = json.load(f) # v2 格式需要转换为 v1 兼容的扁平列表 if is_v2 and isinstance(content_list, list) and content_list and isinstance(content_list[0], list): content_list = _parse_v2_content_list(content_list) logger.info(f"v2 格式已转换为扁平列表,共 {len(content_list)} 项") logger.info(f"读取 content_list: {len(content_list)} 项, 格式={'v2' if is_v2 else 'v1'}") # 读取 Markdown md_path = output_subdir / f"{doc_name}.md" markdown_content = "" if md_path.exists(): with open(md_path, 'r', encoding='utf-8') as f: markdown_content = f.read() # 第一遍扫描:收集所有文本内容(用于构建图片上下文) text_items = [] # [(index, text, page_idx), ...] for idx, item in enumerate(content_list): if item.get("type") == "text": text = item.get("text", "").strip() if text: text_items.append((idx, text, item.get("page_idx", 0))) def get_context_for_image(image_idx: int, page_idx: int, window: int = 3) -> tuple: """获取图片前后的文本上下文""" context_before = [] context_after = [] # 查找图片前后的文本项 for item_idx, text, item_page in text_items: if item_idx < image_idx and item_page >= page_idx - 1: # 图片之前的文本(同页或上一页) context_before.append(text) elif item_idx > image_idx and item_page <= page_idx + 1: # 图片之后的文本(同页或下一页) context_after.append(text) # 只保留最近的 window 条 context_before = context_before[-window:] if context_before else [] context_after = context_after[:window] if context_after else [] return " ".join(context_before), " ".join(context_after) # 解析 content_list for idx, item in enumerate(content_list): item_type = item.get("type", "text") page_idx = item.get("page_idx", 0) bbox = item.get("bbox", []) text_level = item.get("text_level", 0) if item_type == "text": text = item.get("text", "") v2_styles = item.get("_v2_styles", []) # 启发式标题识别(当 text_level 为 0 时) if text_level == 0: from parsers.heading_rules import get_heading_engine engine = get_heading_engine() detected_level, _ = engine.detect(text, style=v2_styles) text_level = detected_level # 处理标题 title = "" if text_level > 0: title = text.strip() # 更新章节栈 while section_stack and section_stack[-1][0] >= text_level: section_stack.pop() section_stack.append((text_level, title)) # 构建章节路径 section_path = " > ".join([s[1] for s in section_stack]) # 构建 Markdown if text_level > 0: md_line = f"{'#' * text_level} {text}" else: md_line = text markdown_parts.append(md_line) chunk = MinerUChunk( content=text, chunk_type="heading" if text_level > 0 else "text", page_start=page_idx + 1, page_end=page_idx + 1, text_level=text_level, title=title, section_path=section_path, bbox=bbox, source_file=file_path.name ) chunks.append(chunk) elif item_type == "table": table_body = item.get("table_body", "") table_caption = item.get("table_caption", "") table_footnote = item.get("_v2_table_footnote", "") # 表格也可能有图片形式(img_path) img_path = item.get("img_path", "") section_path = " > ".join([s[1] for s in section_stack]) markdown_parts.append(f"\n| 表格 |") if table_body: md_table = html_table_to_markdown(table_body) markdown_parts.append(md_table) else: md_table = "" # 从原始 HTML 提取嵌入图片(md_table 经 get_text 转换后已丢失 标签) table_images = extract_images_from_markdown(table_body) if table_body else [] # 表格内容增强:caption + footnote table_content = table_caption or "表格" if table_footnote: table_content = f"{table_content}\n[脚注] {table_footnote}" chunk = MinerUChunk( content=table_content, chunk_type="table", page_start=page_idx + 1, page_end=page_idx + 1, title=table_caption or "表格", section_path=section_path, bbox=bbox, source_file=file_path.name, table_html=table_body, image_path=img_path, # 表格的独立图片形式 images=table_images if table_images else None, # 嵌入图片列表 table_type=item.get("_v2_table_type", ""), table_nest_level=item.get("_v2_table_nest_level", ""), ) chunks.append(chunk) if table_body: tables.append(table_body) # 表格图片也加入 images 列表 if img_path: images.append(img_path) elif item_type in ("image", "chart"): # 处理图片和图表类型(MinerU 将图表识别为 chart 类型) img_path = item.get("img_path", "") caption = item.get("caption", "") vlm_desc = item.get("_v2_vlm_description", "") chart_md = item.get("_v2_chart_markdown", "") sub_type = item.get("sub_type", "") section_path = " > ".join([s[1] for s in section_stack]) markdown_parts.append(f"\n![{caption}]({img_path})") if chart_md: markdown_parts.append(chart_md) # 图表类型标记为 chart,便于后续区分处理 chunk_type = "chart" if item_type == "chart" else "image" # 获取图片上下文 context_before, context_after = get_context_for_image(idx, page_idx) # 图片内容增强:caption + VLM 描述 content_text = caption or ("图表" if item_type == "chart" else "图片") if vlm_desc: content_text = f"{content_text}\n[视觉描述] {vlm_desc}" if chart_md: content_text = f"{content_text}\n[数据表]\n{chart_md}" chunk = MinerUChunk( content=content_text, chunk_type=chunk_type, page_start=page_idx + 1, page_end=page_idx + 1, title=caption or ("图表" if item_type == "chart" else "图片"), section_path=section_path, bbox=bbox, source_file=file_path.name, image_path=img_path, context_before=context_before, context_after=context_after, vlm_description=vlm_desc, chart_markdown=chart_md, table_html=chart_md if chart_md else None, sub_type=sub_type, ) chunks.append(chunk) if img_path: images.append(img_path) if chart_md: tables.append(chart_md) elif item_type == "equation": # 处理公式类型 equation_content = item.get("content", "") or item.get("latex", "") or item.get("text", "") img_path = item.get("img_path", "") section_path = " > ".join([s[1] for s in section_stack]) if equation_content: markdown_parts.append(f"\n$$ {equation_content} $$") chunk = MinerUChunk( content=equation_content or "公式", chunk_type="equation", page_start=page_idx + 1, page_end=page_idx + 1, title="公式", section_path=section_path, bbox=bbox, source_file=file_path.name, image_path=img_path ) chunks.append(chunk) if img_path: images.append(img_path) # 如果没有从 content_list 解析到内容,使用 Markdown if not chunks and markdown_content: markdown_parts = [markdown_content] # 后处理:过滤空切片 → 合并碎片 → 拆分超长 # 使用配置中的切片约束 try: from config import MIN_CHUNK_SIZE, MAX_CHUNK_SIZE min_merge = MIN_CHUNK_SIZE // 2 # 合并阈值为最小切片的一半 max_size = MAX_CHUNK_SIZE except ImportError: min_merge = 100 max_size = 1200 # 表单类型二次校正(在 _post_process_chunks 之前,因为 table 不参与合并) _reclassify_text_chunks(chunks) chunks = _post_process_chunks(chunks, min_merge_size=min_merge, max_chunk_size=max_size) # 验证分类标题是否被规则引擎正确识别(安全网,仅告警不修改) _validate_category_section_paths(chunks) return { 'markdown': "\n".join(markdown_parts), 'chunks': chunks, 'tables': tables, 'images': images, 'content_list': content_list } def _post_process_chunks( chunks: List[MinerUChunk], min_merge_size: int = 100, max_merged_size: int = 800, max_chunk_size: int = 1000 ) -> List[MinerUChunk]: """ 后处理:过滤空切片 → 合并碎片 → 拆分超长 解决三个问题: 1. 空切片(0 字符)入库 2. 标题/短文本独立成片导致碎片化 3. 超长切片超过 Embedding 模型的 token 限制 策略: - 标题 chunk 与下方第一个正文 chunk 合并 - 连续短文本 chunk(< min_merge_size)合并 - 合并后超过 max_merged_size 则停止合并 - 表格、图片 chunk 保持独立不参与合并 - 最终检查:超过 max_chunk_size 的 chunk 使用 split_text_with_limit 拆分 Args: chunks: 原始 chunk 列表 min_merge_size: 短于此长度的 chunk 触发合并 max_merged_size: 合并后的最大字符数 max_chunk_size: 单个 chunk 的硬性上限 Returns: 处理后的 chunk 列表 """ if not chunks: return [] # Phase 1: 过滤空切片(统一处理 list/string 类型) filtered = [] for c in chunks: if not c.content: continue # 统一转字符串 if isinstance(c.content, list): c.content = '\n'.join(str(item) for item in c.content) if c.content.strip(): filtered.append(c) chunks = filtered if not chunks: return [] # Phase 2: 合并碎片 merged = [] buffer = None # 当前合并缓冲 _buffer_has_body = False # 缓冲是否已包含正文(防止标题继续合并) for chunk in chunks: # 表格、图片和图表不参与合并,直接输出 if chunk.chunk_type in ('table', 'image', 'chart', 'equation'): if buffer: merged.append(buffer) buffer = None _buffer_has_body = False merged.append(chunk) continue # 标题 chunk(text_level > 0) if chunk.text_level > 0: if buffer: # H1 级标题是章节边界,强制断开,不参与连续标题链合并 if chunk.text_level == 1: merged.append(buffer) _buffer_has_body = False # 连续标题链合并:缓冲也是纯标题(无正文)时,合并而非刷新(仅非 H1) elif buffer.text_level > 0 and not _buffer_has_body: combined = buffer.content.rstrip() + '\n' + chunk.content if len(combined) <= max_merged_size: buffer.content = combined buffer.page_end = chunk.page_end # 取更高层级(数值更小) buffer.text_level = min(buffer.text_level, chunk.text_level) # section_path 保留第一个(更高级别)的 continue # 合并后超限,刷新缓冲 merged.append(buffer) _buffer_has_body = False else: # 缓冲是正文,正常刷新 merged.append(buffer) _buffer_has_body = False # 标题作为新缓冲的起点 buffer = MinerUChunk( content=chunk.content, chunk_type='text', page_start=chunk.page_start, page_end=chunk.page_end, text_level=chunk.text_level, title=chunk.title, section_path=chunk.section_path, bbox=chunk.bbox, source_file=chunk.source_file, ) _buffer_has_body = False continue # 正文 chunk content_len = len(chunk.content.strip()) if buffer is None: # 没有缓冲,开始新缓冲 if content_len < min_merge_size: buffer = MinerUChunk( content=chunk.content, chunk_type='text', page_start=chunk.page_start, page_end=chunk.page_end, text_level=chunk.text_level, title=chunk.title, section_path=chunk.section_path, bbox=chunk.bbox, source_file=chunk.source_file, ) _buffer_has_body = True # 正文 chunk 创建的缓冲已含正文 else: # 足够长,直接输出 merged.append(chunk) else: # 有缓冲,尝试合并 combined_len = len(buffer.content) + 1 + content_len if combined_len <= max_merged_size: # 合并 buffer.content = buffer.content.rstrip() + '\n' + chunk.content buffer.page_end = chunk.page_end # 正文并入标题缓冲后,标记已含正文,防止后续标题继续合并 # 保留 text_level 不置零,使标题层级信息传递到向量库 _buffer_has_body = True else: # 超过上限,输出缓冲,当前 chunk 开始新缓冲或直接输出 merged.append(buffer) if content_len < min_merge_size: buffer = MinerUChunk( content=chunk.content, chunk_type='text', page_start=chunk.page_start, page_end=chunk.page_end, text_level=chunk.text_level, title=chunk.title, section_path=chunk.section_path, bbox=chunk.bbox, source_file=chunk.source_file, ) _buffer_has_body = True # 正文 chunk 创建的缓冲已含正文 else: buffer = None _buffer_has_body = False merged.append(chunk) # 刷新最后的缓冲 if buffer: merged.append(buffer) # Phase 3: 拆分超长切片 result = [] for chunk in merged: if chunk.chunk_type in ('table', 'image', 'chart', 'equation'): result.append(chunk) continue if len(chunk.content) > max_chunk_size: # 使用已有的 split_text_with_limit 函数拆分 try: from core.chunker import split_text_with_limit sub_texts = split_text_with_limit( chunk.content, chunk_size=max_chunk_size, overlap=50, max_length=max_chunk_size ) for i, sub_text in enumerate(sub_texts): sub_chunk = MinerUChunk( content=sub_text, chunk_type=chunk.chunk_type, page_start=chunk.page_start, page_end=chunk.page_end, text_level=chunk.text_level if i == 0 else 0, title=chunk.title if i == 0 else '', section_path=chunk.section_path, bbox=chunk.bbox, source_file=chunk.source_file, ) result.append(sub_chunk) except ImportError: logger.warning("split_text_with_limit 不可用,保留原始超长切片") result.append(chunk) else: result.append(chunk) logger.info( f"切片后处理: {len(chunks)} → {len(result)} " f"(过滤空切片+合并碎片+拆分超长)" ) return result def _split_oversized_text(text: str, max_size: int) -> List[str]: """ 拆分超长文本,优先在句子边界切分 先尝试使用 core.chunker.split_text_with_limit(如果可用), 否则使用内置的句子边界拆分。 Args: text: 待拆分文本 max_size: 单片最大字符数 Returns: 拆分后的文本列表 """ if len(text) <= max_size: return [text] # 尝试使用 LangChain 分块器 try: from core.chunker import split_text_with_limit result = split_text_with_limit(text, chunk_size=max_size, overlap=50, max_length=max_size) if result: return result except (ImportError, Exception): pass # 内置回退:按句子边界拆分 import re sentences = re.split(r'(?<=[。!?.!?\n])', text) chunks = [] current = "" for sentence in sentences: if not sentence: continue if len(current) + len(sentence) <= max_size: current += sentence else: if current: chunks.append(current) # 单句超长则硬截断 if len(sentence) > max_size: for i in range(0, len(sentence), max_size): chunks.append(sentence[i:i + max_size]) current = "" else: current = sentence if current: chunks.append(current) return chunks if chunks else [text[:max_size]] def convert_to_rag_format( result: Dict[str, Any], source_file: str ) -> List[Dict]: """ 将 MinerU 结果转换为 RAG 入库格式 Args: result: parse_with_mineru() 返回结果 source_file: 源文件名 Returns: [{'text': ..., 'page': ..., 'has_table': ..., ...}, ...] """ pages_content = [] for chunk in result['chunks']: # 跳过空内容 content = chunk.content # content 可能是字符串或列表 if isinstance(content, list): content = '\n'.join(str(item) for item in content) if not content or not content.strip(): continue # 构建内容文本(基于已处理的 content) if chunk.chunk_type == "table" and chunk.table_html: # 表格:将 HTML 转为 Markdown 格式 content = f"【表格】{chunk.title}\n\n{html_table_to_markdown(chunk.table_html)}" elif chunk.chunk_type == "image": content = f"【图片】{chunk.title}" elif chunk.chunk_type == "equation": content = f"【公式】{content}" elif chunk.text_level > 0: # 标题 prefix = "#" * chunk.text_level content = f"{prefix} {content}" page_info = { 'text': content, 'page': chunk.page_start, 'page_end': chunk.page_end, 'has_table': chunk.chunk_type == "table", 'section': chunk.title, 'section_path': chunk.section_path, 'level': chunk.text_level, 'chunk_type': chunk.chunk_type, 'source_file': source_file, 'is_mineru_chunk': True # 标记为 MinerU 输出 } if chunk.bbox: # 确保 bbox 是纯 Python 列表(转换 numpy 类型) page_info['bbox'] = [float(x) for x in chunk.bbox] if chunk.bbox else None pages_content.append(page_info) return pages_content def html_table_to_markdown(html_table: str) -> str: """ 将 HTML 表格转换为 Markdown 格式 处理 rowspan 合并单元格 """ import re from bs4 import BeautifulSoup try: soup = BeautifulSoup(html_table, 'html.parser') table = soup.find('table') if not table: return html_table rows = table.find_all('tr') if not rows: return html_table # 计算最大列数 max_cols = 0 for row in rows: cols_in_row = 0 for cell in row.find_all(['td', 'th']): colspan = int(cell.get('colspan', 1)) cols_in_row += colspan max_cols = max(max_cols, cols_in_row) if max_cols == 0: return html_table # 处理表格,记录 rowspan 占用 rowspan_tracker = {} # {col: remaining_rows} md_rows = [] for row_idx, row in enumerate(rows): cells = [] col_idx = 0 for cell in row.find_all(['td', 'th']): # 跳过被 rowspan 占用的列 while col_idx in rowspan_tracker: cells.append("") # 占位符 rowspan_tracker[col_idx] -= 1 if rowspan_tracker[col_idx] <= 0: del rowspan_tracker[col_idx] col_idx += 1 # 提取单元格内容(保留图片引用信息) img_tags = cell.find_all('img') if img_tags: # 单元格包含图片,生成占位标记供 LLM 感知 text_part = cell.get_text(strip=True) img_count = len(img_tags) if text_part: content = f"{text_part} [{'图片' if img_count == 1 else f'{img_count}张图片'}]" else: content = f"[{'图片' if img_count == 1 else f'{img_count}张图片'}]" else: content = cell.get_text(strip=True) # 处理 rowspan rowspan = int(cell.get('rowspan', 1)) colspan = int(cell.get('colspan', 1)) # 当前单元格 cells.append(content) # 处理 colspan for _ in range(colspan - 1): cells.append("") col_idx += 1 # 记录 rowspan(当前列,不是+1后的列) if rowspan > 1: rowspan_tracker[col_idx] = rowspan - 1 col_idx += 1 # 补齐剩余列(被 rowspan 占用的列) while col_idx < max_cols: if col_idx in rowspan_tracker: cells.append("") rowspan_tracker[col_idx] -= 1 if rowspan_tracker[col_idx] <= 0: del rowspan_tracker[col_idx] else: cells.append("") col_idx += 1 # 构建 Markdown 行 if cells: md_row = "| " + " | ".join(cells) + " |" md_rows.append(md_row) # 第一行后添加分隔线 if row_idx == 0: separator = "| " + " | ".join(["---"] * len(cells)) + " |" md_rows.append(separator) return "\n".join(md_rows) except Exception: # 如果 BeautifulSoup 解析失败,回退到简单实现 rows = re.findall(r']*>(.*?)', html_table, re.DOTALL) if not rows: return html_table md_rows = [] for i, row in enumerate(rows): cells = re.findall(r']*>(.*?)', row, re.DOTALL) cells = [re.sub(r'<[^>]+>', '', cell).strip() for cell in cells] if cells: md_row = "| " + " | ".join(cells) + " |" md_rows.append(md_row) if i == 0: separator = "| " + " | ".join(["---"] * len(cells)) + " |" md_rows.append(separator) return "\n".join(md_rows) # ========== 工具函数 ========== def compute_file_hash(file_path: str) -> str: """ 计算文件 MD5 Hash,用于隔离输出目录 Args: file_path: 文件路径 Returns: 12 位 hash 字符串 """ hash_md5 = hashlib.md5() with open(file_path, 'rb') as f: for chunk in iter(lambda: f.read(8192), b''): hash_md5.update(chunk) return hash_md5.hexdigest()[:12] def parse_with_mineru_persistent( file_path: str, output_base: str = ".data/mineru_temp", images_output: str = ".data/images", lang: str = "ch", enable_table: bool = True, enable_formula: bool = True, backend: str = None, start_page: int = 0, end_page: int = 99999, cleanup_after_image_move: bool = True ) -> Dict[str, Any]: """ 使用 MinerU 解析文档(扁平化存储) 用于分步流水线架构: - Step 1 (parse.py): 本地 GPU 运行 MinerU,输出持久化 - Step 2 (embed.py): 读取 JSON,调用远端 API 生成摘要/描述 Args: file_path: 文档文件路径 output_base: MinerU 临时输出目录,默认 .data/mineru_temp(解析后自动清理) images_output: 图片存储目录,默认 .data/images(扁平化) lang: 语言代码 (ch, en, etc.) enable_table: 启用表格识别 enable_formula: 启用公式识别 backend: 解析后端 ("pipeline", "vlm-auto-engine", "hybrid-auto-engine") start_page: 起始页码(0-indexed,仅 PDF 有效) end_page: 结束页码(仅 PDF 有效) cleanup_after_image_move: 图片移动后是否清理临时输出(默认 True) Returns: { 'markdown': str, # Markdown 内容 'chunks': List[MinerUChunk], # 结构化分块 'tables': List[str], # 表格列表 'images': List[str], # 图片列表(已更新为最终路径) 'content_list': List[Dict],# 原始 content_list 'output_dir': str, # MinerU 输出目录 'file_hash': str # 文件 hash } """ file_path = Path(file_path) if not file_path.exists(): raise FileNotFoundError(f"文件不存在: {file_path}") # 从 config 读取默认 backend if backend is None: try: from config import MINERU_LOCAL_BACKEND backend = MINERU_LOCAL_BACKEND except ImportError: backend = 'pipeline' # 检查文件大小 file_size = file_path.stat().st_size if file_size > MAX_PDF_SIZE: raise ValueError(f"文件过大: {file_size / 1024 / 1024:.1f}MB,最大允许 {MAX_PDF_SIZE / 1024 / 1024:.0f}MB") # 计算文件 hash,用于隔离输出目录 file_hash = compute_file_hash(str(file_path)) output_dir = Path(output_base) / file_hash # 清理已存在的输出目录,避免权限冲突 if output_dir.exists(): shutil.rmtree(output_dir, ignore_errors=True) # 检查是否优先使用在线 API from config import MINERU_PREFER_ONLINE, MINERU_API_TOKEN use_online = MINERU_PREFER_ONLINE and MINERU_API_TOKEN if use_online: # 优先使用在线 API try: logger.info(f"优先使用 MinerU 在线 API 解析: {file_path.name}") result = parse_with_mineru_online(str(file_path)) except Exception as e: logger.warning(f"在线 API 解析失败,回退到本地: {e}") use_online = False if not use_online: # 使用本地 MinerU result = parse_with_mineru( file_path=str(file_path), output_dir=str(output_dir), lang=lang, enable_table=enable_table, enable_formula=enable_formula, backend=backend, start_page=start_page, end_page=end_page ) # 移动图片到统一存储目录 # 即使 result['images'] 为空,也要检查 images 目录(处理嵌入表格的图片) images_dir = Path(images_output) images_dir.mkdir(parents=True, exist_ok=True) # 获取 MinerU 输出的图片源目录 doc_name = file_path.stem # 检查是否是在线解析(有 _zip_content 字段) zip_content = result.pop('_zip_content', None) zip_images = result.pop('_zip_images', []) if zip_content: # 在线解析:从 zip 包提取图片到临时目录 import zipfile import io img_src_dir = output_dir / doc_name / "images" img_src_dir.mkdir(parents=True, exist_ok=True) # 提取 zip 中的所有图片(使用 zip_images 列表) extracted_count = 0 with zipfile.ZipFile(io.BytesIO(zip_content), 'r') as zf: for img_path in zip_images: try: with zf.open(img_path) as img_file: img_content = img_file.read() img_name = os.path.basename(img_path) img_dst = img_src_dir / img_name with open(img_dst, 'wb') as f: f.write(img_content) extracted_count += 1 logger.debug(f"提取图片: {img_path} -> {img_dst}") except KeyError: logger.warning(f"zip 中找不到图片: {img_path}") logger.info(f"从 zip 包提取 {extracted_count} 张图片到: {img_src_dir}") else: # 本地解析:使用 MinerU 输出目录 auto_dir = output_dir / doc_name / "auto" office_dir = output_dir / doc_name / "office" if auto_dir.exists(): img_src_dir = auto_dir / "images" elif office_dir.exists(): img_src_dir = office_dir / "images" else: img_src_dir = None # 图片路径映射:旧路径 -> 新文件名 image_path_map = {} # { "images/abc.jpg": "0569dd285537.jpg" } # 首先处理 content_list.json 中引用的图片 for img_path in result['images']: # img_path 是相对路径,如 "images/abc123.jpg" if img_src_dir: # 直接使用 img_src_dir 作为图片目录 img_name = os.path.basename(img_path) src_path = img_src_dir / img_name logger.debug(f"检查图片: img_path={img_path}, src_path={src_path}, exists={src_path.exists()}") if src_path.exists(): # 计算 hash 避免冲突 img_hash = compute_file_hash(str(src_path)) new_name = f"{img_hash}{src_path.suffix}" dst_path = images_dir / new_name # 移动图片 shutil.move(str(src_path), str(dst_path)) # 记录映射:旧路径 -> 新文件名(只存文件名,不含目录) image_path_map[img_path] = new_name logger.debug(f"移动图片: {src_path} -> {dst_path}") else: # 源文件不存在,记录警告 logger.warning(f"图片源文件不存在: {src_path} (引用路径: {img_path})") else: logger.warning(f"图片源目录不存在,无法移动: {img_path}") # 扫描 images 目录中所有剩余图片(处理不在 content_list.json 中的图片) # 这些图片可能是嵌入表格中的图片,MinerU 没有在 content_list.json 中记录 if img_src_dir and img_src_dir.exists(): for img_file in img_src_dir.iterdir(): if img_file.suffix.lower() in {'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.webp'}: # 计算 hash 并移动 img_hash = compute_file_hash(str(img_file)) new_name = f"{img_hash}{img_file.suffix}" dst_path = images_dir / new_name shutil.move(str(img_file), str(dst_path)) # 记录映射(使用相对路径作为 key) rel_path = f"images/{img_file.name}" image_path_map[rel_path] = new_name logger.debug(f"移动额外图片: {img_file} -> {dst_path}") # 更新 chunks 中的图片路径(单一来源:只在这里写路径) for chunk in result['chunks']: # 更新所有类型切片的 image_path(包括 table) if hasattr(chunk, 'image_path') and chunk.image_path: # 使用映射表更新为新文件名 if chunk.image_path in image_path_map: chunk.image_path = image_path_map[chunk.image_path] else: # 兼容:尝试直接匹配文件名 basename = os.path.basename(chunk.image_path) for old_path, new_name in image_path_map.items(): if basename in old_path: chunk.image_path = new_name break # 更新表格嵌入图片的路径映射(images 字段) if hasattr(chunk, 'images') and chunk.images: for img_info in chunk.images: if isinstance(img_info, dict) and 'id' in img_info: old_id = img_info['id'] if old_id in image_path_map: img_info['id'] = image_path_map[old_id] else: # 兼容:尝试用文件名匹配映射 old_basename = os.path.basename(old_id) for old_path, new_name in image_path_map.items(): if old_basename == os.path.basename(old_path): img_info['id'] = new_name break # 更新结果中的图片路径列表(供外部使用) result['images'] = list(image_path_map.values()) # 清理 MinerU 输出目录(如果请求) if cleanup_after_image_move and output_dir.exists(): shutil.rmtree(output_dir, ignore_errors=True) logger.info(f"已清理 MinerU 输出目录: {output_dir}") # 添加额外元数据 result['output_dir'] = str(output_dir) result['file_hash'] = file_hash return result # ========== 格式特定别名 ========== def parse_pdf_with_mineru(*args, **kwargs) -> Dict[str, Any]: """PDF 解析别名""" return parse_with_mineru(*args, **kwargs) def parse_docx_with_mineru(*args, **kwargs) -> Dict[str, Any]: """Word 文档解析别名""" return parse_with_mineru(*args, **kwargs) def parse_xlsx_with_mineru(*args, **kwargs) -> Dict[str, Any]: """Excel 解析别名""" return parse_with_mineru(*args, **kwargs) def parse_pptx_with_mineru(*args, **kwargs) -> Dict[str, Any]: """PowerPoint 解析别名""" return parse_with_mineru(*args, **kwargs) # ========== 兼容性别名 ========== # 提供与旧 pdf_odl 模块兼容的接口 ChunkMetadata = MinerUChunk if __name__ == "__main__": import sys if sys.platform == 'win32': sys.stdout.reconfigure(encoding='utf-8') if len(sys.argv) < 2: print("用法: python mineru_parser.py <文件路径>") print(f"支持格式: {', '.join(SUPPORTED_FORMATS.keys())}") sys.exit(1) file_path = sys.argv[1] print(f"正在解析: {file_path}") result = parse_with_mineru(file_path) print(f"\n解析完成:") print(f"- Markdown 长度: {len(result['markdown'])} 字符") print(f"- 分块数量: {len(result['chunks'])}") print(f"- 表格数量: {len(result['tables'])}") print(f"- 图片数量: {len(result['images'])}") # 显示前几个分块 print("\n前 5 个分块:") for i, chunk in enumerate(result['chunks'][:5]): print(f"\n--- Chunk {i+1} ({chunk.chunk_type}) ---") print(f"页码: {chunk.page_start}") print(f"标题: {chunk.title}") print(f"章节: {chunk.section_path}") print(chunk.content[:100] + "..." if len(chunk.content) > 100 else chunk.content)