"""一次性清洗:修正历史知识笔记里遗留的 `## Related` 结构链接与 `### Sources`。 背景 ==== 早期沉淀的每条笔记都会在 `## Related` 段自动塞入“结构性” wikilink (`[[index]]` / `[[_meta/taxonomy]]` / `[[projects//…]]` / 项目 hub), 并在 `## Related` 下嵌套一个 `### Sources` 子段。现在: - 真·相关链接只用裸标题 `[[Title]]`(由 service 层按共享标签注入); - 来源统一由结构化的「来源」区块(`note.sources`)展示,不再写进正文。 但历史笔记的 `content_md`(数据库 + vault 文件)里仍焊死着这些遗留内容, 导致详情页出现噪声链接、以及 sources/来源 重复。本脚本把它们清理掉。 清洗规则(保守,只删明确匹配项) ================================ 对每条笔记的正文: 1. 删除 `## Related` 段内的 `### Sources` 子段(标题 + 其下条目); 2. 删除 `## Related` 段内的“结构性”链接条目 —— 即 wikilink 目标包含 `/`、 或等于 `index` / `taxonomy` 之类的脚手架页(真·相关链接是裸标题,予以保留); 3. 若清理后 `## Related` 段已无任何内容条目,则连同 `## Related` 标题一并删除; 4. 折叠多余空行。 裸标题的 `[[Title]]` 相关链接、其它正文内容一律保留(无数据丢失)。 用法 ==== :: # 预览(不写盘),打印每条将要发生的改动 PYTHONPATH=. python scripts/cleanup_knowledge_related.py --dry-run # 执行(写回数据库 + 重写 vault 文件) PYTHONPATH=. python scripts/cleanup_knowledge_related.py # 只处理单条 PYTHONPATH=. python scripts/cleanup_knowledge_related.py --note-id kb_xxx 脚本幂等:已经干净的笔记会被跳过。写回走 `KnowledgeService.update_note`, 所以每次改动都会落一份版本快照(可在「编辑历史」回滚)。 """ from __future__ import annotations import argparse import asyncio import logging import re import sys logger = logging.getLogger("cleanup_knowledge_related") _RELATED_HEADER_RE = re.compile(r"^##\s+(Related|相关知识|相关文档|相关)\s*$", re.IGNORECASE) _SOURCES_HEADER_RE = re.compile(r"^###\s+(Sources|来源)\s*$", re.IGNORECASE) _H12_RE = re.compile(r"^#{1,2}\s+") _H123_RE = re.compile(r"^#{1,3}\s+") _BULLET_LINK_RE = re.compile(r"^[-*]\s*\[\[([^\]\n|]+)(?:\|[^\]\n]+)?\]\]\s*$") def _is_structural_target(target: str) -> bool: """A `## Related` wikilink that points at vault scaffolding (not a real note). Real related cross-links are bare note titles (`[[标题]]`) with no slashes. The legacy structural links were full vault paths / meta pages. """ t = (target or "").strip().strip("/").lower() if not t: return True if "/" in t: # projects/

/…, _meta/taxonomy, … — all structural return True return t in {"index", "taxonomy", "hot", "log", "_meta"} def clean_note_body(md: str) -> str: """Strip the legacy `### Sources` subsection + structural `## Related` links. Pure function (no I/O) so it can be unit-tested. Returns the cleaned body; identical to the input when there is nothing to clean. """ lines = md.split("\n") n = len(lines) out: list[str] = [] i = 0 in_fence = False while i < n: line = lines[i] if re.match(r"^\s*(```|~~~)", line): in_fence = not in_fence out.append(line) i += 1 continue if in_fence or not _RELATED_HEADER_RE.match(line.strip()): out.append(line) i += 1 continue # --- inside a `## Related` section ------------------------------- header = line j = i + 1 kept: list[str] = [] # surviving content lines of this section while j < n: cur = lines[j] stripped = cur.strip() if _H12_RE.match(cur): # next h1/h2 ends the Related section break if _SOURCES_HEADER_RE.match(stripped): # Drop the `### Sources` header and everything until the next header. j += 1 while j < n and not _H123_RE.match(lines[j]): j += 1 continue m = _BULLET_LINK_RE.match(stripped) if m and _is_structural_target(m.group(1)): j += 1 # drop the structural link bullet continue kept.append(cur) j += 1 meaningful = [ln for ln in kept if ln.strip()] if meaningful: out.append(header) out.extend(kept) # else: drop the now-empty `## Related` section entirely. i = j cleaned = "\n".join(out) cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).rstrip() + "\n" return cleaned async def _run(dry_run: bool, only_note_id: str | None) -> int: from deerflow.config import get_app_config from deerflow.knowledge import make_knowledge_service from deerflow.persistence.engine import close_engine, get_session_factory, init_engine_from_config config = get_app_config() await init_engine_from_config(config.database) sf = get_session_factory() if sf is None: logger.error("No database session factory — is the database backend configured?") return 1 service = make_knowledge_service(sf, config.knowledge) if service is None: logger.error("Knowledge base disabled (knowledge.enabled=false) or memory backend; nothing to do.") return 1 notes = await service.repo.list_all(include_archived=True) if only_note_id: notes = [n for n in notes if n.get("id") == only_note_id] scanned = changed = 0 for note in notes: scanned += 1 body = note.get("content_md") or "" cleaned = clean_note_body(body) if cleaned == body: continue changed += 1 nid = note.get("id") title = note.get("title") or nid if dry_run: logger.info("[dry-run] would clean %s — %s (%d → %d chars)", nid, title, len(body), len(cleaned)) else: await service.update_note(nid, content_md=cleaned, updated_by="cleanup_script") logger.info("cleaned %s — %s", nid, title) await close_engine() logger.info("Done. scanned=%d, %s=%d", scanned, "would-change" if dry_run else "changed", changed) return 0 def main() -> int: logging.basicConfig(level=logging.INFO, format="%(message)s") parser = argparse.ArgumentParser(description="Clean legacy `## Related` structural links + `### Sources` from knowledge notes.") parser.add_argument("--dry-run", action="store_true", help="Preview changes without writing.") parser.add_argument("--note-id", default=None, help="Only process this note id.") args = parser.parse_args() return asyncio.run(_run(args.dry_run, args.note_id)) if __name__ == "__main__": sys.exit(main())