From d141baf9643ecf991a15306c4ecd00e9517d056e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=88=98=E8=88=AA=E5=AE=87?= <3364451258@qq.com> Date: Sun, 5 Jul 2026 04:46:47 +0800 Subject: [PATCH] fix: use path hash prefix for source_file names to prevent same-name collisions; add size filter to ingest script --- ingest_progress.txt | 56 +++++++++++++++++++++++++++++++ ingest_stdout.txt | 57 +++++++++++++++++++++++++++++++ scripts/ingest_obsidian.py | 69 ++++++++++++++++++++++++++++++++++++++ src/core/ingest.py | 13 +++++-- 4 files changed, 192 insertions(+), 3 deletions(-) create mode 100644 ingest_progress.txt create mode 100644 ingest_stdout.txt create mode 100644 scripts/ingest_obsidian.py diff --git a/ingest_progress.txt b/ingest_progress.txt new file mode 100644 index 0000000..7f31f0a --- /dev/null +++ b/ingest_progress.txt @@ -0,0 +1,56 @@ +初始化... +待处理: 46 个文件 +跳过超大: 6 个 + SKIP 人工智能协会内容创作通知.md (1915KB) + SKIP Git团队协作指南.md (94KB) + SKIP Git团队协作指南.sync-conflict-20260615-122209-R4RFYM7.md (94KB) + SKIP OpenClaw介绍.md (64KB) + SKIP Git内部原理详解.md (59KB) + SKIP Git团队协作指南大纲.md (58KB) +[1/46] 17 chunks | 博客/AI时代的Token战争-能源算力与我们每个人的选择.md (0.6s) +[2/46] 39 chunks | 博客/AI时代的VibeCoding与程序员进化.md (1.2s) +[3/46] 59 chunks | 博客/CLI在AI时代的浴火重生.md (1.7s) +[4/46] 68 chunks | 博客/DeepSeek-V4全面解析.md (1.1s) +[5/46] 16 chunks | 博客/大模型赋能架构设计.md (0.5s) +[6/46] 27 chunks | 博客/大语言模型算子详解.md (1.0s) +[7/46] 8 chunks | 博客/学习Agent,越学越像在重新理解操作系统.md (0.3s) +[8/46] 69 chunks | 博客/深入解析 Claude Code:Vibe Coding 时代的 AI 编程利器.md (1.5s) +[9/46] 12 chunks | 博客/自建Git服务-在NAS上部署Gitea.md (0.4s) +[10/46] 38 chunks | 博客/DeepSeek-V4博客大纲.md (0.4s) +[11/46] 110 chunks | 博客/uv工具推荐博客大纲.md (1.0s) +[12/46] 48 chunks | 博客/AI助你轻松上手LaTeX论文写作.md (1.1s) +[13/46] 15 chunks | 博客/SEO分析报告.md (0.4s) +[14/46] 18 chunks | 博客/博客爬取报告.md (0.3s) +[15/46] 15 chunks | 博客/大数据技术栈.md (0.3s) +[16/46] 7 chunks | 博客/摘要检查报告.md (0.3s) +[17/46] 27 chunks | 博客/文章列表.md (0.5s) +[18/46] 19 chunks | 博客/标签分类审查报告.md (0.7s) +[19/46] 13 chunks | 博客/LinearRegression线性回归.md (0.6s) +[20/46] 53 chunks | 博客/从全连接层到卷积.md (1.2s) +[21/46] 114 chunks | 博客/深度学习完全指南.md (1.6s) +[22/46] 43 chunks | 博客/视觉语言模型技术综述.md (1.4s) +[23/46] 52 chunks | 博客/Docker安装与入门指南.md (1.4s) +[24/46] 167 chunks | 博客/Docker部署完全指南.md (3.6s) +[25/46] 115 chunks | 博客/Git团队协作指南(精简版).md (1.3s) +[26/46] 10 chunks | 博客/OpenClaw安装教程.md (0.1s) +[27/46] 247 chunks | 博客/uv工具推荐博客.md (2.8s) +[28/46] 6 chunks | Club/CLAUDE.md (0.1s) +[29/46] 7 chunks | Club/README.md (0.1s) +[30/46] 11 chunks | Club/仓库说明.md (0.4s) +[31/46] 9 chunks | Club/导航页.md (0.3s) +[32/46] 3 chunks | Club/说明.md (0.1s) +[33/46] 4 chunks | Club/说明.md (0.2s) +[34/46] 4 chunks | Club/MetaRL核心团队实施方案.md (0.3s) +[35/46] 3 chunks | Club/说明.md (0.1s) +[36/46] 20 chunks | Club/定量考核材料清单.md (0.7s) +[37/46] 14 chunks | Club/年度工作总结起草指南.md (0.6s) +[38/46] 3 chunks | Club/说明.md (0.2s) +[39/46] 3 chunks | Club/说明.md (0.1s) +[40/46] 7 chunks | Club/人工智能协会近期工作任务清单.md (0.3s) +[41/46] 3 chunks | Club/说明.md (0.2s) +[42/46] 7 chunks | halo/README.md (0.3s) +[43/46] 6 chunks | halo/README.zh-CN.md (0.3s) +[44/46] 21 chunks | halo/usage-guide.md (0.4s) +[45/46] 9 chunks | 顶层/CLAUDE.md (0.4s) +[46/46] 57 chunks | 顶层/temp-agent-os-v2.md (1.6s) +[DONE] 完成: 46 文件, 1623 chunks (45s) diff --git a/ingest_stdout.txt b/ingest_stdout.txt new file mode 100644 index 0000000..4b23c91 --- /dev/null +++ b/ingest_stdout.txt @@ -0,0 +1,57 @@ +ʼ... + Loading weights: 0%| | 0/71 [00:0050KB CPU嵌入太慢).""" +import sys, time +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).parent.parent / "src")) +from src.core.config import load_config +from src.core.db import VectorDB +from src.core.embedder import create_embedder +from src.core.ingest import DocumentIngestor + +MAX_SIZE = 50_000 # 跳过 >50KB 的文件 + +progress_file = Path(__file__).parent.parent / "ingest_progress.txt" +def log(msg): + print(msg, flush=True) + with open(progress_file, "a", encoding="utf-8") as f: + f.write(msg + "\n") + +t0 = time.time() +log("初始化...") +cfg = load_config() +db = VectorDB(persist_dir=cfg.chroma.persist_dir) +embedder = create_embedder(cfg.embed) +ingestor = DocumentIngestor(db, embedder, "obsidian_blog") + +targets = [ + ("博客", "D:/Code/Obsidian/博客"), + ("Club", "D:/Code/Obsidian/Club-Service-Guide"), + ("halo", "D:/Code/Obsidian/obsidian-halo"), +] +files = [] +skipped = [] +for label, d in targets: + if not Path(d).exists(): continue + for f in Path(d).rglob("*.md"): + if any(p.startswith(".") for p in f.parts): continue + if "node_modules" in f.parts: continue + size = f.stat().st_size + if size > MAX_SIZE: + skipped.append((f.name, size)) + continue + files.append((label, str(f))) +for f in Path("D:/Code/Obsidian").glob("*.md"): + sz = f.stat().st_size + if sz > MAX_SIZE: + skipped.append((f.name, sz)) + else: + files.append(("顶层", str(f))) + +log(f"待处理: {len(files)} 个文件") +if skipped: + log(f"跳过超大: {len(skipped)} 个") + for name, sz in sorted(skipped, key=lambda x: -x[1]): + log(f" SKIP {name} ({sz//1024}KB)") + +total = 0 +for i, (label, fp) in enumerate(files, 1): + name = Path(fp).name + t1 = time.time() + try: + n = ingestor.ingest_file(fp) + total += n + dt = time.time() - t1 + log(f"[{i}/{len(files)}] {n:>4d} chunks | {label}/{name} ({dt:.1f}s)") + except Exception as e: + log(f"[{i}/{len(files)}] ERROR {label}/{name}: {e}") + +elapsed = time.time() - t0 +log(f"[DONE] 完成: {len(files)} 文件, {total} chunks ({elapsed:.0f}s)") diff --git a/src/core/ingest.py b/src/core/ingest.py index b7a14b7..a9e7d58 100644 --- a/src/core/ingest.py +++ b/src/core/ingest.py @@ -173,10 +173,17 @@ class DocumentIngestor: return self.db.get_or_create_collection(self.collection_name) def ingest_file(self, file_path: str) -> int: - """入库单个 Markdown 文件, 返回 chunk 数量.""" - path = Path(file_path) + """入库单个 Markdown 文件, 返回 chunk 数量. + + 使用文件路径的 SHA256 前 12 位 + 文件名作为唯一标识, + 避免不同目录下同名文件冲突. + """ + import hashlib + path = Path(file_path).resolve() content = path.read_text(encoding="utf-8") - file_name = path.name + # 用路径 hash 保证同名文件在不同目录下不冲突 + path_hash = hashlib.sha256(str(path).encode()).hexdigest()[:12] + file_name = f"{path_hash}_{path.name}" return self.ingest_content(content, file_name)