refactor: 将核心代码迁移到 scripts 目录

- 创建 scripts/ 目录作为核心代码根目录 - 移动 core/, readers/, utils/ 到 scripts/ 下 - 移动 config.py, lyxy_document_reader.py 到 scripts/ - 移动 encoding_detection.py 到 scripts/utils/ - 更新 pyproject.toml 中的入口点路径和 pytest 配置 - 更新所有内部导入语句为 scripts.* 模块 - 更新 README.md 目录结构说明 - 更新 openspec/config.yaml 添加目录结构说明 - 删除无用的 main.py 此变更使项目结构更清晰，便于区分核心代码与测试、文档等支撑文件。
2026-03-08 17:41:03 +08:00
parent 750ef50a8d
commit 15b63800a8
50 changed files with 66 additions and 60 deletions
--- a/scripts/readers/pdf/init.py
+++ b/scripts/readers/pdf/init.py
@@ -0,0 +1,57 @@
+"""PDF 文件阅读器，支持多种解析方法（OCR 优先）。"""
+
+import os
+from typing import List, Optional, Tuple
+
+from scripts.readers.base import BaseReader
+from scripts.utils import is_valid_pdf
+
+from . import docling_ocr
+from . import unstructured_ocr
+from . import docling
+from . import unstructured
+from . import markitdown
+from . import pypdf
+
+
+PARSERS = [
+    ("docling OCR", docling_ocr.parse),
+    ("unstructured OCR", unstructured_ocr.parse),
+    ("docling", docling.parse),
+    ("unstructured", unstructured.parse),
+    ("MarkItDown", markitdown.parse),
+    ("pypdf", pypdf.parse),
+]
+
+
+class PdfReader(BaseReader):
+    """PDF 文件阅读器"""
+
+    @property
+    def supported_extensions(self) -> List[str]:
+        return [".pdf"]
+
+    def supports(self, file_path: str) -> bool:
+        return file_path.endswith('.pdf')
+
+    def parse(self, file_path: str) -> Tuple[Optional[str], List[str]]:
+        failures = []
+
+        # 检查文件是否存在
+        if not os.path.exists(file_path):
+            return None, ["文件不存在"]
+
+        # 验证文件格式
+        if not is_valid_pdf(file_path):
+            return None, ["不是有效的 PDF 文件"]
+
+        content = None
+
+        for parser_name, parser_func in PARSERS:
+            content, error = parser_func(file_path)
+            if content is not None:
+                return content, failures
+            else:
+                failures.append(f"- {parser_name}: {error}")
+
+        return None, failures