183 lines
6.8 KiB
Python
183 lines
6.8 KiB
Python
"""一次性清洗:修正历史知识笔记里遗留的 `## Related` 结构链接与 `### Sources`。
|
|
|
|
背景
|
|
====
|
|
早期沉淀的每条笔记都会在 `## Related` 段自动塞入“结构性” wikilink
|
|
(`[[index]]` / `[[_meta/taxonomy]]` / `[[projects/<project>/…]]` / 项目 hub),
|
|
并在 `## Related` 下嵌套一个 `### Sources` 子段。现在:
|
|
|
|
- 真·相关链接只用裸标题 `[[Title]]`(由 service 层按共享标签注入);
|
|
- 来源统一由结构化的「来源」区块(`note.sources`)展示,不再写进正文。
|
|
|
|
但历史笔记的 `content_md`(数据库 + vault 文件)里仍焊死着这些遗留内容,
|
|
导致详情页出现噪声链接、以及 sources/来源 重复。本脚本把它们清理掉。
|
|
|
|
清洗规则(保守,只删明确匹配项)
|
|
================================
|
|
对每条笔记的正文:
|
|
1. 删除 `## Related` 段内的 `### Sources` 子段(标题 + 其下条目);
|
|
2. 删除 `## Related` 段内的“结构性”链接条目 —— 即 wikilink 目标包含 `/`、
|
|
或等于 `index` / `taxonomy` 之类的脚手架页(真·相关链接是裸标题,予以保留);
|
|
3. 若清理后 `## Related` 段已无任何内容条目,则连同 `## Related` 标题一并删除;
|
|
4. 折叠多余空行。
|
|
|
|
裸标题的 `[[Title]]` 相关链接、其它正文内容一律保留(无数据丢失)。
|
|
|
|
用法
|
|
====
|
|
::
|
|
|
|
# 预览(不写盘),打印每条将要发生的改动
|
|
PYTHONPATH=. python scripts/cleanup_knowledge_related.py --dry-run
|
|
|
|
# 执行(写回数据库 + 重写 vault 文件)
|
|
PYTHONPATH=. python scripts/cleanup_knowledge_related.py
|
|
|
|
# 只处理单条
|
|
PYTHONPATH=. python scripts/cleanup_knowledge_related.py --note-id kb_xxx
|
|
|
|
脚本幂等:已经干净的笔记会被跳过。写回走 `KnowledgeService.update_note`,
|
|
所以每次改动都会落一份版本快照(可在「编辑历史」回滚)。
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import asyncio
|
|
import logging
|
|
import re
|
|
import sys
|
|
|
|
logger = logging.getLogger("cleanup_knowledge_related")
|
|
|
|
_RELATED_HEADER_RE = re.compile(r"^##\s+(Related|相关知识|相关文档|相关)\s*$", re.IGNORECASE)
|
|
_SOURCES_HEADER_RE = re.compile(r"^###\s+(Sources|来源)\s*$", re.IGNORECASE)
|
|
_H12_RE = re.compile(r"^#{1,2}\s+")
|
|
_H123_RE = re.compile(r"^#{1,3}\s+")
|
|
_BULLET_LINK_RE = re.compile(r"^[-*]\s*\[\[([^\]\n|]+)(?:\|[^\]\n]+)?\]\]\s*$")
|
|
|
|
|
|
def _is_structural_target(target: str) -> bool:
|
|
"""A `## Related` wikilink that points at vault scaffolding (not a real note).
|
|
|
|
Real related cross-links are bare note titles (`[[标题]]`) with no slashes.
|
|
The legacy structural links were full vault paths / meta pages.
|
|
"""
|
|
t = (target or "").strip().strip("/").lower()
|
|
if not t:
|
|
return True
|
|
if "/" in t: # projects/<p>/…, _meta/taxonomy, … — all structural
|
|
return True
|
|
return t in {"index", "taxonomy", "hot", "log", "_meta"}
|
|
|
|
|
|
def clean_note_body(md: str) -> str:
|
|
"""Strip the legacy `### Sources` subsection + structural `## Related` links.
|
|
|
|
Pure function (no I/O) so it can be unit-tested. Returns the cleaned body;
|
|
identical to the input when there is nothing to clean.
|
|
"""
|
|
lines = md.split("\n")
|
|
n = len(lines)
|
|
out: list[str] = []
|
|
i = 0
|
|
in_fence = False
|
|
while i < n:
|
|
line = lines[i]
|
|
if re.match(r"^\s*(```|~~~)", line):
|
|
in_fence = not in_fence
|
|
out.append(line)
|
|
i += 1
|
|
continue
|
|
if in_fence or not _RELATED_HEADER_RE.match(line.strip()):
|
|
out.append(line)
|
|
i += 1
|
|
continue
|
|
|
|
# --- inside a `## Related` section -------------------------------
|
|
header = line
|
|
j = i + 1
|
|
kept: list[str] = [] # surviving content lines of this section
|
|
while j < n:
|
|
cur = lines[j]
|
|
stripped = cur.strip()
|
|
if _H12_RE.match(cur): # next h1/h2 ends the Related section
|
|
break
|
|
if _SOURCES_HEADER_RE.match(stripped):
|
|
# Drop the `### Sources` header and everything until the next header.
|
|
j += 1
|
|
while j < n and not _H123_RE.match(lines[j]):
|
|
j += 1
|
|
continue
|
|
m = _BULLET_LINK_RE.match(stripped)
|
|
if m and _is_structural_target(m.group(1)):
|
|
j += 1 # drop the structural link bullet
|
|
continue
|
|
kept.append(cur)
|
|
j += 1
|
|
|
|
meaningful = [ln for ln in kept if ln.strip()]
|
|
if meaningful:
|
|
out.append(header)
|
|
out.extend(kept)
|
|
# else: drop the now-empty `## Related` section entirely.
|
|
i = j
|
|
|
|
cleaned = "\n".join(out)
|
|
cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).rstrip() + "\n"
|
|
return cleaned
|
|
|
|
|
|
async def _run(dry_run: bool, only_note_id: str | None) -> int:
|
|
from deerflow.config import get_app_config
|
|
from deerflow.knowledge import make_knowledge_service
|
|
from deerflow.persistence.engine import close_engine, get_session_factory, init_engine_from_config
|
|
|
|
config = get_app_config()
|
|
await init_engine_from_config(config.database)
|
|
sf = get_session_factory()
|
|
if sf is None:
|
|
logger.error("No database session factory — is the database backend configured?")
|
|
return 1
|
|
service = make_knowledge_service(sf, config.knowledge)
|
|
if service is None:
|
|
logger.error("Knowledge base disabled (knowledge.enabled=false) or memory backend; nothing to do.")
|
|
return 1
|
|
|
|
notes = await service.repo.list_all(include_archived=True)
|
|
if only_note_id:
|
|
notes = [n for n in notes if n.get("id") == only_note_id]
|
|
|
|
scanned = changed = 0
|
|
for note in notes:
|
|
scanned += 1
|
|
body = note.get("content_md") or ""
|
|
cleaned = clean_note_body(body)
|
|
if cleaned == body:
|
|
continue
|
|
changed += 1
|
|
nid = note.get("id")
|
|
title = note.get("title") or nid
|
|
if dry_run:
|
|
logger.info("[dry-run] would clean %s — %s (%d → %d chars)", nid, title, len(body), len(cleaned))
|
|
else:
|
|
await service.update_note(nid, content_md=cleaned, updated_by="cleanup_script")
|
|
logger.info("cleaned %s — %s", nid, title)
|
|
|
|
await close_engine()
|
|
logger.info("Done. scanned=%d, %s=%d", scanned, "would-change" if dry_run else "changed", changed)
|
|
return 0
|
|
|
|
|
|
def main() -> int:
|
|
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
parser = argparse.ArgumentParser(description="Clean legacy `## Related` structural links + `### Sources` from knowledge notes.")
|
|
parser.add_argument("--dry-run", action="store_true", help="Preview changes without writing.")
|
|
parser.add_argument("--note-id", default=None, help="Only process this note id.")
|
|
args = parser.parse_args()
|
|
return asyncio.run(_run(args.dry_run, args.note_id))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|