feat: 添加 doc/xls/ppt 旧格式文档支持

- 新增 DocReader，支持 markitdown 和 pypandoc-binary 解析器 - 新增 XlsReader，支持 unstructured、markitdown 和 pandas+xlrd 解析器 - 新增 PptReader，支持 markitdown 解析器 - 添加 olefile 依赖用于验证 OLE2 格式 - 更新 config.py 添加 doc/xls/ppt 依赖配置 - 更新 --advice 支持 doc/xls/ppt 格式 - 添加相应的测试用例 - 同步 specs 到主目录
2026-03-10 23:09:13 +08:00
parent e53e64d386
commit cf10458dd6
32 changed files with 756 additions and 5 deletions
--- a/scripts/config.py
+++ b/scripts/config.py
@@ -97,5 +97,37 @@ DEPENDENCIES = {
                "selenium"
            ]
        }
+    },
+    "doc": {
+        "default": {
+            "python": None,
+            "dependencies": [
+                "markitdown[docx]",
+                "pypandoc-binary",
+                "olefile"
+            ]
+        }
+    },
+    "xls": {
+        "default": {
+            "python": None,
+            "dependencies": [
+                "unstructured[xlsx]",
+                "markitdown[xls]",
+                "pandas",
+                "tabulate",
+                "xlrd",
+                "olefile"
+            ]
+        }
+    },
+    "ppt": {
+        "default": {
+            "python": None,
+            "dependencies": [
+                "markitdown[pptx]",
+                "olefile"
+            ]
+        }
    }
 }
--- a/scripts/core/advice_generator.py
+++ b/scripts/core/advice_generator.py
@@ -12,6 +12,9 @@ from readers import (
    XlsxReader,
    PptxReader,
    HtmlReader,
+    DocReader,
+    XlsReader,
+    PptReader,
 )


@@ -22,6 +25,9 @@ _READER_KEY_MAP: Dict[Type[BaseReader], str] = {
    XlsxReader: "xlsx",
    PptxReader: "pptx",
    HtmlReader: "html",
+    DocReader: "doc",
+    XlsReader: "xls",
+    PptReader: "ppt",
 }


--- a/scripts/lyxy_document_reader.py
+++ b/scripts/lyxy_document_reader.py
@@ -1,5 +1,5 @@
 #!/usr/bin/env python3
-"""文档解析器命令行交互模块，提供命令行接口。支持 DOCX、PPTX、XLSX、PDF、HTML 和 URL。"""
+"""文档解析器命令行交互模块，提供命令行接口。支持 DOC、DOCX、XLS、XLSX、PPT、PPTX、PDF、HTML 和 URL。"""

 import argparse
 import logging
@@ -39,10 +39,10 @@ from readers import READERS

 def main() -> None:
    parser = argparse.ArgumentParser(
-        description="将 DOCX、PPTX、XLSX、PDF、HTML 文件或 URL 解析为 Markdown"
+        description="将 DOC、DOCX、XLS、XLSX、PPT、PPTX、PDF、HTML 文件或 URL 解析为 Markdown"
    )

-    parser.add_argument("input_path", help="DOCX、PPTX、XLSX、PDF、HTML 文件或 URL")
+    parser.add_argument("input_path", help="DOC、DOCX、XLS、XLSX、PPT、PPTX、PDF、HTML 文件或 URL")

    parser.add_argument(
        "-a",
--- a/scripts/readers/init.py
+++ b/scripts/readers/init.py
@@ -6,6 +6,9 @@ from .xlsx import XlsxReader
 from .pptx import PptxReader
 from .pdf import PdfReader
 from .html import HtmlReader
+from .doc import DocReader
+from .xls import XlsReader
+from .ppt import PptReader

 READERS = [
    DocxReader,
@@ -13,6 +16,9 @@ READERS = [
    PptxReader,
    PdfReader,
    HtmlReader,
+    DocReader,
+    XlsReader,
+    PptReader,
 ]

 __all__ = [
@@ -22,5 +28,8 @@ __all__ = [
    "PptxReader",
    "PdfReader",
    "HtmlReader",
+    "DocReader",
+    "XlsReader",
+    "PptReader",
    "READERS",
 ]
--- a/scripts/readers/doc/init.py
+++ b/scripts/readers/doc/init.py
@@ -0,0 +1,48 @@
+"""DOC 文件阅读器，支持多种解析方法。"""
+
+import os
+from typing import List, Optional, Tuple
+
+from readers.base import BaseReader
+from utils import is_valid_doc
+
+from . import markitdown
+from . import pypandoc
+
+
+PARSERS = [
+    ("MarkItDown", markitdown.parse),
+    ("pypandoc-binary", pypandoc.parse),
+]
+
+
+class DocReader(BaseReader):
+    """DOC 文件阅读器"""
+
+    def supports(self, file_path: str) -> bool:
+        return file_path.lower().endswith('.doc')
+
+    def parse(self, file_path: str) -> Tuple[Optional[str], List[str]]:
+        failures = []
+
+        # 检查文件是否存在
+        if not os.path.exists(file_path):
+            return None, ["文件不存在"]
+
+        # 验证文件格式
+        if not is_valid_doc(file_path):
+            return None, ["不是有效的 DOC 文件"]
+
+        content = None
+
+        for parser_name, parser_func in PARSERS:
+            try:
+                content, error = parser_func(file_path)
+                if content is not None:
+                    return content, failures
+                else:
+                    failures.append(f"- {parser_name}: {error}")
+            except Exception as e:
+                failures.append(f"- {parser_name}: [意外异常] {type(e).__name__}: {str(e)}")
+
+        return None, failures
--- a/scripts/readers/doc/markitdown.py
+++ b/scripts/readers/doc/markitdown.py
@@ -0,0 +1,10 @@
+"""使用 MarkItDown 库解析 DOC 文件"""
+
+from typing import Optional, Tuple
+
+from readers._utils import parse_via_markitdown
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 MarkItDown 库解析 DOC 文件"""
+    return parse_via_markitdown(file_path)
--- a/scripts/readers/doc/pypandoc.py
+++ b/scripts/readers/doc/pypandoc.py
@@ -0,0 +1,29 @@
+"""使用 pypandoc-binary 库解析 DOC 文件"""
+
+from typing import Optional, Tuple
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 pypandoc-binary 库解析 DOC 文件"""
+    try:
+        import pypandoc
+    except ImportError:
+        return None, "pypandoc-binary 库未安装"
+
+    try:
+        content = pypandoc.convert_file(
+            source_file=file_path,
+            to="md",
+            format="doc",
+            outputfile=None,
+            extra_args=["--wrap=none"],
+        )
+    except OSError as exc:
+        return None, f"pypandoc-binary 缺少 Pandoc 可执行文件: {exc}"
+    except RuntimeError as exc:
+        return None, f"pypandoc-binary 解析失败: {exc}"
+
+    content = content.strip()
+    if not content:
+        return None, "文档为空"
+    return content, None
--- a/scripts/readers/ppt/init.py
+++ b/scripts/readers/ppt/init.py
@@ -0,0 +1,46 @@
+"""PPT 文件阅读器，支持多种解析方法。"""
+
+import os
+from typing import List, Optional, Tuple
+
+from readers.base import BaseReader
+from utils import is_valid_ppt
+
+from . import markitdown
+
+
+PARSERS = [
+    ("MarkItDown", markitdown.parse),
+]
+
+
+class PptReader(BaseReader):
+    """PPT 文件阅读器"""
+
+    def supports(self, file_path: str) -> bool:
+        return file_path.lower().endswith('.ppt')
+
+    def parse(self, file_path: str) -> Tuple[Optional[str], List[str]]:
+        failures = []
+
+        # 检查文件是否存在
+        if not os.path.exists(file_path):
+            return None, ["文件不存在"]
+
+        # 验证文件格式
+        if not is_valid_ppt(file_path):
+            return None, ["不是有效的 PPT 文件"]
+
+        content = None
+
+        for parser_name, parser_func in PARSERS:
+            try:
+                content, error = parser_func(file_path)
+                if content is not None:
+                    return content, failures
+                else:
+                    failures.append(f"- {parser_name}: {error}")
+            except Exception as e:
+                failures.append(f"- {parser_name}: [意外异常] {type(e).__name__}: {str(e)}")
+
+        return None, failures
--- a/scripts/readers/ppt/markitdown.py
+++ b/scripts/readers/ppt/markitdown.py
@@ -0,0 +1,10 @@
+"""使用 MarkItDown 库解析 PPT 文件"""
+
+from typing import Optional, Tuple
+
+from readers._utils import parse_via_markitdown
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 MarkItDown 库解析 PPT 文件"""
+    return parse_via_markitdown(file_path)
--- a/scripts/readers/xls/init.py
+++ b/scripts/readers/xls/init.py
@@ -0,0 +1,50 @@
+"""XLS 文件阅读器，支持多种解析方法。"""
+
+import os
+from typing import List, Optional, Tuple
+
+from readers.base import BaseReader
+from utils import is_valid_xls
+
+from . import unstructured
+from . import markitdown
+from . import pandas
+
+
+PARSERS = [
+    ("unstructured", unstructured.parse),
+    ("MarkItDown", markitdown.parse),
+    ("pandas+xlrd", pandas.parse),
+]
+
+
+class XlsReader(BaseReader):
+    """XLS 文件阅读器"""
+
+    def supports(self, file_path: str) -> bool:
+        return file_path.lower().endswith('.xls')
+
+    def parse(self, file_path: str) -> Tuple[Optional[str], List[str]]:
+        failures = []
+
+        # 检查文件是否存在
+        if not os.path.exists(file_path):
+            return None, ["文件不存在"]
+
+        # 验证文件格式
+        if not is_valid_xls(file_path):
+            return None, ["不是有效的 XLS 文件"]
+
+        content = None
+
+        for parser_name, parser_func in PARSERS:
+            try:
+                content, error = parser_func(file_path)
+                if content is not None:
+                    return content, failures
+                else:
+                    failures.append(f"- {parser_name}: {error}")
+            except Exception as e:
+                failures.append(f"- {parser_name}: [意外异常] {type(e).__name__}: {str(e)}")
+
+        return None, failures
--- a/scripts/readers/xls/markitdown.py
+++ b/scripts/readers/xls/markitdown.py
@@ -0,0 +1,10 @@
+"""使用 MarkItDown 库解析 XLS 文件"""
+
+from typing import Optional, Tuple
+
+from readers._utils import parse_via_markitdown
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 MarkItDown 库解析 XLS 文件"""
+    return parse_via_markitdown(file_path)
--- a/scripts/readers/xls/pandas.py
+++ b/scripts/readers/xls/pandas.py
@@ -0,0 +1,41 @@
+"""使用 pandas+xlrd 库解析 XLS 文件"""
+
+from typing import Optional, Tuple
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 pandas+xlrd 库解析 XLS 文件"""
+    try:
+        import pandas as pd
+        from tabulate import tabulate
+    except ImportError as e:
+        if "pandas" in str(e):
+            missing_lib = "pandas"
+        elif "xlrd" in str(e):
+            missing_lib = "xlrd"
+        else:
+            missing_lib = "tabulate"
+        return None, f"{missing_lib} 库未安装"
+
+    try:
+        sheets = pd.read_excel(file_path, sheet_name=None, engine="xlrd")
+
+        markdown_parts = []
+        for sheet_name, df in sheets.items():
+            if len(df) == 0:
+                markdown_parts.append(f"## {sheet_name}\n\n*工作表为空*")
+                continue
+
+            table_md = tabulate(
+                df, headers="keys", tablefmt="pipe", showindex=True, missingval=""
+            )
+            markdown_parts.append(f"## {sheet_name}\n\n{table_md}")
+
+        if not markdown_parts:
+            return None, "Excel 文件为空"
+
+        markdown_content = "# Excel数据转换结果\n\n" + "\n\n".join(markdown_parts)
+
+        return markdown_content, None
+    except Exception as e:
+        return None, f"pandas 解析失败: {str(e)}"
--- a/scripts/readers/xls/unstructured.py
+++ b/scripts/readers/xls/unstructured.py
@@ -0,0 +1,22 @@
+"""使用 unstructured 库解析 XLS 文件"""
+
+from typing import Optional, Tuple
+
+from readers._utils import convert_unstructured_to_markdown
+
+
+def parse(file_path: str) -> Tuple[Optional[str], Optional[str]]:
+    """使用 unstructured 库解析 XLS 文件"""
+    try:
+        from unstructured.partition.xlsx import partition_xlsx
+    except ImportError:
+        return None, "unstructured 库未安装"
+
+    try:
+        elements = partition_xlsx(filename=file_path, infer_table_structure=True)
+        content = convert_unstructured_to_markdown(elements)
+        if not content.strip():
+            return None, "文档为空"
+        return content, None
+    except Exception as e:
+        return None, f"unstructured 解析失败: {str(e)}"
--- a/scripts/utils/init.py
+++ b/scripts/utils/init.py
@@ -5,6 +5,9 @@ from .file_detection import (
    is_valid_pptx,
    is_valid_xlsx,
    is_valid_pdf,
+    is_valid_doc,
+    is_valid_xls,
+    is_valid_ppt,
    is_html_file,
    is_url,
    detect_file_type,
@@ -15,6 +18,9 @@ __all__ = [
    "is_valid_pptx",
    "is_valid_xlsx",
    "is_valid_pdf",
+    "is_valid_doc",
+    "is_valid_xls",
+    "is_valid_ppt",
    "is_html_file",
    "is_url",
    "detect_file_type",
--- a/scripts/utils/file_detection.py
+++ b/scripts/utils/file_detection.py
@@ -5,6 +5,19 @@ import zipfile
 from typing import List, Optional


+def _is_valid_ole(file_path: str) -> bool:
+    """验证 OLE2 格式文件（DOC/XLS/PPT）"""
+    try:
+        import olefile
+    except ImportError:
+        # 如果 olefile 未安装，就不做严格验证
+        return True
+    try:
+        return olefile.isOleFile(file_path)
+    except Exception:
+        return False
+
+
 def _is_valid_ooxml(file_path: str, required_files: List[str]) -> bool:
    """验证 OOXML 格式文件（DOCX/PPTX/XLSX）"""
    try:
@@ -35,6 +48,21 @@ def is_valid_xlsx(file_path: str) -> bool:
    return _is_valid_ooxml(file_path, _XLSX_REQUIRED)


+def is_valid_doc(file_path: str) -> bool:
+    """验证文件是否为有效的 DOC 格式"""
+    return _is_valid_ole(file_path)
+
+
+def is_valid_xls(file_path: str) -> bool:
+    """验证文件是否为有效的 XLS 格式"""
+    return _is_valid_ole(file_path)
+
+
+def is_valid_ppt(file_path: str) -> bool:
+    """验证文件是否为有效的 PPT 格式"""
+    return _is_valid_ole(file_path)
+
+
 def is_valid_pdf(file_path: str) -> bool:
    """验证文件是否为有效的 PDF 格式"""
    try:
@@ -61,13 +89,18 @@ _FILE_TYPE_VALIDATORS = {
    ".pptx": is_valid_pptx,
    ".xlsx": is_valid_xlsx,
    ".pdf": is_valid_pdf,
+    ".doc": is_valid_doc,
+    ".xls": is_valid_xls,
+    ".ppt": is_valid_ppt,
 }


 def detect_file_type(file_path: str) -> Optional[str]:
-    """检测文件类型，返回 'docx'、'pptx'、'xlsx' 或 'pdf'"""
+    """检测文件类型，返回 'docx'、'pptx'、'xlsx'、'pdf'、'doc'、'xls' 或 'ppt'"""
    ext = os.path.splitext(file_path)[1].lower()
    validator = _FILE_TYPE_VALIDATORS.get(ext)
    if validator and validator(file_path):
        return ext.lstrip(".")
    return None
+
+