deerflow-code/offline-backend-20260512/backend/scripts/cleanup_knowledge_related.py
2026-09-07 18:24:55 +08:00

183 lines
6.8 KiB
Python

"""一次性清洗:修正历史知识笔记里遗留的 `## Related` 结构链接与 `### Sources`。
背景
====
早期沉淀的每条笔记都会在 `## Related` 段自动塞入“结构性” wikilink
(`[[index]]` / `[[_meta/taxonomy]]` / `[[projects/<project>/…]]` / 项目 hub),
并在 `## Related` 下嵌套一个 `### Sources` 子段。现在:
- 真·相关链接只用裸标题 `[[Title]]`(由 service 层按共享标签注入);
- 来源统一由结构化的「来源」区块(`note.sources`)展示,不再写进正文。
但历史笔记的 `content_md`(数据库 + vault 文件)里仍焊死着这些遗留内容,
导致详情页出现噪声链接、以及 sources/来源 重复。本脚本把它们清理掉。
清洗规则(保守,只删明确匹配项)
================================
对每条笔记的正文:
1. 删除 `## Related` 段内的 `### Sources` 子段(标题 + 其下条目);
2. 删除 `## Related` 段内的“结构性”链接条目 —— 即 wikilink 目标包含 `/`、
或等于 `index` / `taxonomy` 之类的脚手架页(真·相关链接是裸标题,予以保留);
3. 若清理后 `## Related` 段已无任何内容条目,则连同 `## Related` 标题一并删除;
4. 折叠多余空行。
裸标题的 `[[Title]]` 相关链接、其它正文内容一律保留(无数据丢失)。
用法
====
::
# 预览(不写盘),打印每条将要发生的改动
PYTHONPATH=. python scripts/cleanup_knowledge_related.py --dry-run
# 执行(写回数据库 + 重写 vault 文件)
PYTHONPATH=. python scripts/cleanup_knowledge_related.py
# 只处理单条
PYTHONPATH=. python scripts/cleanup_knowledge_related.py --note-id kb_xxx
脚本幂等:已经干净的笔记会被跳过。写回走 `KnowledgeService.update_note`,
所以每次改动都会落一份版本快照(可在「编辑历史」回滚)。
"""
from __future__ import annotations
import argparse
import asyncio
import logging
import re
import sys
logger = logging.getLogger("cleanup_knowledge_related")
_RELATED_HEADER_RE = re.compile(r"^##\s+(Related|相关知识|相关文档|相关)\s*$", re.IGNORECASE)
_SOURCES_HEADER_RE = re.compile(r"^###\s+(Sources|来源)\s*$", re.IGNORECASE)
_H12_RE = re.compile(r"^#{1,2}\s+")
_H123_RE = re.compile(r"^#{1,3}\s+")
_BULLET_LINK_RE = re.compile(r"^[-*]\s*\[\[([^\]\n|]+)(?:\|[^\]\n]+)?\]\]\s*$")
def _is_structural_target(target: str) -> bool:
"""A `## Related` wikilink that points at vault scaffolding (not a real note).
Real related cross-links are bare note titles (`[[标题]]`) with no slashes.
The legacy structural links were full vault paths / meta pages.
"""
t = (target or "").strip().strip("/").lower()
if not t:
return True
if "/" in t: # projects/<p>/…, _meta/taxonomy, … — all structural
return True
return t in {"index", "taxonomy", "hot", "log", "_meta"}
def clean_note_body(md: str) -> str:
"""Strip the legacy `### Sources` subsection + structural `## Related` links.
Pure function (no I/O) so it can be unit-tested. Returns the cleaned body;
identical to the input when there is nothing to clean.
"""
lines = md.split("\n")
n = len(lines)
out: list[str] = []
i = 0
in_fence = False
while i < n:
line = lines[i]
if re.match(r"^\s*(```|~~~)", line):
in_fence = not in_fence
out.append(line)
i += 1
continue
if in_fence or not _RELATED_HEADER_RE.match(line.strip()):
out.append(line)
i += 1
continue
# --- inside a `## Related` section -------------------------------
header = line
j = i + 1
kept: list[str] = [] # surviving content lines of this section
while j < n:
cur = lines[j]
stripped = cur.strip()
if _H12_RE.match(cur): # next h1/h2 ends the Related section
break
if _SOURCES_HEADER_RE.match(stripped):
# Drop the `### Sources` header and everything until the next header.
j += 1
while j < n and not _H123_RE.match(lines[j]):
j += 1
continue
m = _BULLET_LINK_RE.match(stripped)
if m and _is_structural_target(m.group(1)):
j += 1 # drop the structural link bullet
continue
kept.append(cur)
j += 1
meaningful = [ln for ln in kept if ln.strip()]
if meaningful:
out.append(header)
out.extend(kept)
# else: drop the now-empty `## Related` section entirely.
i = j
cleaned = "\n".join(out)
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).rstrip() + "\n"
return cleaned
async def _run(dry_run: bool, only_note_id: str | None) -> int:
from deerflow.config import get_app_config
from deerflow.knowledge import make_knowledge_service
from deerflow.persistence.engine import close_engine, get_session_factory, init_engine_from_config
config = get_app_config()
await init_engine_from_config(config.database)
sf = get_session_factory()
if sf is None:
logger.error("No database session factory — is the database backend configured?")
return 1
service = make_knowledge_service(sf, config.knowledge)
if service is None:
logger.error("Knowledge base disabled (knowledge.enabled=false) or memory backend; nothing to do.")
return 1
notes = await service.repo.list_all(include_archived=True)
if only_note_id:
notes = [n for n in notes if n.get("id") == only_note_id]
scanned = changed = 0
for note in notes:
scanned += 1
body = note.get("content_md") or ""
cleaned = clean_note_body(body)
if cleaned == body:
continue
changed += 1
nid = note.get("id")
title = note.get("title") or nid
if dry_run:
logger.info("[dry-run] would clean %s — %s (%d → %d chars)", nid, title, len(body), len(cleaned))
else:
await service.update_note(nid, content_md=cleaned, updated_by="cleanup_script")
logger.info("cleaned %s — %s", nid, title)
await close_engine()
logger.info("Done. scanned=%d, %s=%d", scanned, "would-change" if dry_run else "changed", changed)
return 0
def main() -> int:
logging.basicConfig(level=logging.INFO, format="%(message)s")
parser = argparse.ArgumentParser(description="Clean legacy `## Related` structural links + `### Sources` from knowledge notes.")
parser.add_argument("--dry-run", action="store_true", help="Preview changes without writing.")
parser.add_argument("--note-id", default=None, help="Only process this note id.")
args = parser.parse_args()
return asyncio.run(_run(args.dry_run, args.note_id))
if __name__ == "__main__":
sys.exit(main())