refactor: 将核心代码迁移到 scripts 目录

- 创建 scripts/ 目录作为核心代码根目录 - 移动 core/, readers/, utils/ 到 scripts/ 下 - 移动 config.py, lyxy_document_reader.py 到 scripts/ - 移动 encoding_detection.py 到 scripts/utils/ - 更新 pyproject.toml 中的入口点路径和 pytest 配置 - 更新所有内部导入语句为 scripts.* 模块 - 更新 README.md 目录结构说明 - 更新 openspec/config.yaml 添加目录结构说明 - 删除无用的 main.py 此变更使项目结构更清晰，便于区分核心代码与测试、文档等支撑文件。
2026-03-08 17:41:03 +08:00
parent 750ef50a8d
commit 15b63800a8
50 changed files with 66 additions and 60 deletions
--- a/scripts/readers/html/init.py
+++ b/scripts/readers/html/init.py
@@ -0,0 +1,89 @@
+"""HTML/URL 文件阅读器，支持多种解析方法。"""
+
+import os
+from typing import List, Optional, Tuple
+
+from scripts.readers.base import BaseReader
+from scripts.utils import is_url
+from scripts.utils import encoding_detection
+
+from . import cleaner
+from . import downloader
+from . import trafilatura
+from . import domscribe
+from . import markitdown
+from . import html2text
+
+
+PARSERS = [
+    ("trafilatura", lambda c, t: trafilatura.parse(c)),
+    ("domscribe", lambda c, t: domscribe.parse(c)),
+    ("MarkItDown", lambda c, t: markitdown.parse(c, t)),
+    ("html2text", lambda c, t: html2text.parse(c)),
+]
+
+
+class HtmlReader(BaseReader):
+    """HTML/URL 文件阅读器"""
+
+    @property
+    def supported_extensions(self) -> List[str]:
+        return [".html", ".htm"]
+
+    def supports(self, file_path: str) -> bool:
+        return is_url(file_path) or file_path.endswith(('.html', '.htm'))
+
+    def download_and_parse(self, url: str) -> Tuple[Optional[str], List[str]]:
+        """下载 URL 并解析"""
+        all_failures = []
+
+        # 下载 HTML
+        html_content, download_failures = downloader.download_html(url)
+        all_failures.extend(download_failures)
+
+        if html_content is None:
+            return None, all_failures
+
+        # 清理 HTML
+        html_content = cleaner.clean_html_content(html_content)
+
+        # 解析 HTML
+        content, parse_failures = self._parse_html_content(html_content, None)
+        all_failures.extend(parse_failures)
+
+        return content, all_failures
+
+    def _parse_html_content(self, html_content: str, temp_file_path: Optional[str]) -> Tuple[Optional[str], List[str]]:
+        """解析 HTML 内容"""
+        failures = []
+        content = None
+
+        for parser_name, parser_func in PARSERS:
+            content, error = parser_func(html_content, temp_file_path)
+            if content is not None:
+                return content, failures
+            else:
+                failures.append(f"- {parser_name}: {error}")
+
+        return None, failures
+
+    def parse(self, file_path: str) -> Tuple[Optional[str], List[str]]:
+        all_failures = []
+
+        # 判断输入类型
+        if is_url(file_path):
+            return self.download_and_parse(file_path)
+
+        # 读取本地 HTML 文件，使用编码检测
+        html_content, error = encoding_detection.read_text_file(file_path)
+        if error:
+            return None, [f"- {error}"]
+
+        # 清理 HTML
+        html_content = cleaner.clean_html_content(html_content)
+
+        # 解析 HTML
+        content, parse_failures = self._parse_html_content(html_content, file_path)
+        all_failures.extend(parse_failures)
+
+        return content, all_failures