feat: 统一文档解析器项目 - 迁移 lyxy-reader-office 和 lyxy-reader-html

## 功能特性 - 建立统一的项目结构，包含 core/、readers/、utils/、tests/ 模块 - 迁移 lyxy-reader-office 的所有解析器（docx、xlsx、pptx、pdf） - 迁移 lyxy-reader-html 的所有解析器（html、url 下载） - 统一 CLI 入口为 lyxy_document_reader.py - 统一 Markdown 后处理逻辑 - 按文件类型组织 readers，每个解析器独立文件 - 依赖分组按文件类型细分（docx、xlsx、pptx、pdf、html、http） - PDF OCR 解析器优先，无参数控制 - 使用 logging 模块替代简单 print - 设计完整的单元测试结构 - 重写项目文档 ## 新增目录/文件 - core/ - 核心模块（异常体系、Markdown 工具、解析调度器） - readers/ - 格式阅读器（base.py + docx/xlsx/pptx/pdf/html） - utils/ - 工具函数（文件类型检测） - tests/ - 测试（conftest.py + test_core/ + test_readers/ + test_utils/） - lyxy_document_reader.py - 统一 CLI 入口 ## 依赖分组 - docx - DOCX 文档解析支持 - xlsx - XLSX 文档解析支持 - pptx - PPTX 文档解析支持 - pdf - PDF 文档解析支持（含 OCR） - html - HTML/URL 解析支持 - http - HTTP/URL 下载支持 - office - Office 格式组合（docx/xlsx/pptx/pdf） - web - Web 格式组合（html/http） - full - 完整功能 - dev - 开发依赖
2026-03-08 13:46:37 +08:00
parent eb8973495e
commit 833018d451
66 changed files with 4054 additions and 0 deletions
--- a/readers/pptx/python_pptx.py
+++ b/readers/pptx/python_pptx.py
@@ -0,0 +1,127 @@
+"""使用 python-pptx 库解析 PPTX 文件"""
+
+from typing import Any, List, Optional, Tuple
+
+from core import build_markdown_table, flush_list_stack
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 python-pptx 库解析 PPTX 文件"""
+    try:
+        from pptx import Presentation
+        from pptx.enum.shapes import MSO_SHAPE_TYPE
+    except ImportError:
+        return None, "python-pptx 库未安装"
+
+    _A_NS = {"a": "http://schemas.openxmlformats.org/drawingml/2006/main"}
+
+    def extract_formatted_text(runs: List[Any]) -> str:
+        """从 PPTX 文本运行中提取带有格式的文本"""
+        result = []
+        for run in runs:
+            if not run.text:
+                continue
+
+            text = run.text
+
+            font = run.font
+            is_bold = getattr(font, "bold", False) or False
+            is_italic = getattr(font, "italic", False) or False
+
+            if is_bold and is_italic:
+                text = f"***{text}***"
+            elif is_bold:
+                text = f"**{text}**"
+            elif is_italic:
+                text = f"*{text}*"
+
+            result.append(text)
+
+        return "".join(result).strip()
+
+    def convert_table_to_md(table: Any) -> str:
+        """将 PPTX 表格转换为 Markdown 格式"""
+        rows_data = []
+        for row in table.rows:
+            row_data = []
+            for cell in row.cells:
+                cell_content = []
+                for para in cell.text_frame.paragraphs:
+                    text = extract_formatted_text(para.runs)
+                    if text:
+                        cell_content.append(text)
+                cell_text = " ".join(cell_content).strip()
+                row_data.append(cell_text if cell_text else "")
+            rows_data.append(row_data)
+        return build_markdown_table(rows_data)
+
+    try:
+        prs = Presentation(file_path)
+        md_content = []
+
+        for slide_num, slide in enumerate(prs.slides, 1):
+            md_content.append(f"\n## Slide {slide_num}\n")
+
+            list_stack = []
+
+            for shape in slide.shapes:
+                if shape.shape_type == MSO_SHAPE_TYPE.PICTURE:
+                    continue
+
+                if hasattr(shape, "has_table") and shape.has_table:
+                    if list_stack:
+                        flush_list_stack(list_stack, md_content)
+
+                    table_md = convert_table_to_md(shape.table)
+                    md_content.append(table_md)
+
+                if hasattr(shape, "text_frame"):
+                    for para in shape.text_frame.paragraphs:
+                        pPr = para._element.pPr
+                        is_list = False
+                        if pPr is not None:
+                            is_list = (
+                                para.level > 0
+                                or pPr.find(".//a:buChar", namespaces=_A_NS) is not None
+                                or pPr.find(".//a:buAutoNum", namespaces=_A_NS) is not None
+                            )
+
+                        if is_list:
+                            level = para.level
+
+                            while len(list_stack) <= level:
+                                list_stack.append("")
+
+                            text = extract_formatted_text(para.runs)
+                            if text:
+                                is_ordered = (
+                                    pPr is not None
+                                    and pPr.find(".//a:buAutoNum", namespaces=_A_NS) is not None
+                                )
+                                marker = "1. " if is_ordered else "- "
+                                indent = "  " * level
+                                list_stack[level] = f"{indent}{marker}{text}"
+
+                                for i in range(len(list_stack)):
+                                    if list_stack[i]:
+                                        md_content.append(list_stack[i] + "\n")
+                                        list_stack[i] = ""
+                        else:
+                            if list_stack:
+                                flush_list_stack(list_stack, md_content)
+
+                            text = extract_formatted_text(para.runs)
+                            if text:
+                                md_content.append(f"{text}\n")
+
+            if list_stack:
+                flush_list_stack(list_stack, md_content)
+
+            md_content.append("---\n")
+
+        content = "\n".join(md_content)
+        if not content.strip():
+            return None, "文档为空"
+        return content, None
+    except Exception as e:
+        return None, f"python-pptx 解析失败: {str(e)}"