diff --git a/src/core/splitters/html.py b/src/core/splitters/html.py new file mode 100644 index 0000000..97dfc3d --- /dev/null +++ b/src/core/splitters/html.py @@ -0,0 +1,44 @@ +"""HTML 文档分块器 — 使用 BeautifulSoup 去标签后委托 TextSplitter.""" +import logging +from src.core.splitters.text import TextSplitter + +logger = logging.getLogger("md-vector-db") + + +class HTMLSplitter: + """HTML 分块器:bs4 去标签提取文字 → TextSplitter 分块. + + 实现 Splitter Protocol,内部组合 TextSplitter 实例。 + """ + + def __init__(self, max_size: int = 1000, overlap: int = 100): + self._text_splitter = TextSplitter(max_size=max_size, overlap=overlap) + + def split(self, text: str, source_file: str = "") -> list[dict]: + """从 HTML 文本去标签并分块. + + Args: + text: HTML 文本内容 + source_file: 来源文件名 + """ + try: + from bs4 import BeautifulSoup + except ImportError: + raise ImportError( + "HTML 支持需要 beautifulsoup4 库. 请执行: uv sync --extra html" + ) + + try: + soup = BeautifulSoup(text, "html.parser") + # 移除 script/style 标签,避免 JS/CSS 内容混入 + for tag in soup(["script", "style"]): + tag.decompose() + plain_text = soup.get_text(separator="\n") + except Exception as e: + logger.error("HTML 解析失败: %s — %s", source_file, e) + raise ValueError(f"HTML 解析失败: {e}") from e + + if not plain_text.strip(): + return [] + + return self._text_splitter.split(plain_text, source_file=source_file)