deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/knowledge/markdown_vault.py
2026-09-07 18:24:55 +08:00

214 lines
8.3 KiB
Python

"""obsidian-wiki compatible Markdown vault writer/reader.
Responsibilities (dev doc §10 ``markdown_vault.py``):
- Generate obsidian-wiki compatible frontmatter.
- Produce safe, path-traversal-proof file names/paths.
- Write/update/read Markdown note files.
- Parse frontmatter + extract ``[[wikilinks]]``.
The vault layout follows §8 of the dev doc and stays compatible with
``ar9av/obsidian-wiki`` (concepts/ references/ synthesis/ journal/ projects/…,
plus index.md / log.md / hot.md / .manifest.json / _meta/taxonomy.md /
wiki-export/).
"""
from __future__ import annotations
import re
import unicodedata
from datetime import datetime
from pathlib import Path
from typing import Any
import yaml
_WIKILINK_RE = re.compile(r"\[\[([^\]|]+)(?:\|[^\]]+)?\]\]")
_FRONTMATTER_RE = re.compile(r"^---\s*\n(.*?)\n---\s*\n?(.*)$", re.DOTALL)
_UNSAFE_CHARS_RE = re.compile(r"[^\w一-鿿-]+")
# Top-level vault directories that must exist (obsidian-wiki native layout).
_VAULT_DIRS = ("concepts", "entities", "skills", "references", "synthesis", "journal", "projects", "_meta", "wiki-export")
def slugify(title: str, *, max_len: int = 60) -> str:
"""Return a filesystem-safe slug (ASCII + CJK), never empty."""
text = unicodedata.normalize("NFKC", (title or "").strip()).lower()
text = text.replace(" ", "-")
text = _UNSAFE_CHARS_RE.sub("-", text).strip("-")
text = re.sub(r"-{2,}", "-", text)
if len(text) > max_len:
text = text[:max_len].rstrip("-")
return text or "note"
def extract_wikilinks(text: str) -> list[str]:
"""Return the targets of every ``[[wikilink]]`` in ``text`` (de-duplicated)."""
seen: list[str] = []
for match in _WIKILINK_RE.finditer(text or ""):
target = match.group(1).strip()
if target and target not in seen:
seen.append(target)
return seen
def normalize_wikilink_target(target: str) -> str:
"""Canonicalize a ``[[wikilink]]`` target for resolution/comparison.
Drops any ``|alias`` part, normalizes slashes, strips a leading ``/`` and a
trailing ``.md`` — so ``[[index]]``, ``index.md`` and ``/index`` all map to
the same canonical ``index`` key used by the index/graph/resolver layers.
"""
t = (target or "").strip()
if not t:
return ""
t = t.split("|", 1)[0].strip()
t = t.replace("\\", "/").lstrip("/")
if t.lower().endswith(".md"):
t = t[:-3]
return t
class VaultPathError(ValueError):
"""Raised when a requested vault path escapes the vault root."""
class MarkdownVault:
"""Reads/writes Markdown files inside a single vault root directory."""
def __init__(self, root: Path | str) -> None:
self.root = Path(root).resolve()
# -- path safety ------------------------------------------------------
def resolve(self, rel_path: str) -> Path:
"""Resolve ``rel_path`` under the vault root, blocking traversal."""
rel = str(rel_path).replace("\\", "/").lstrip("/")
candidate = (self.root / rel).resolve()
try:
candidate.relative_to(self.root)
except ValueError as exc:
raise VaultPathError(f"Path escapes vault root: {rel_path!r}") from exc
return candidate
def note_relpath(self, *, category: str, project_name: str | None, slug: str, note_id: str) -> str:
"""Build a vault-relative path for a note.
Project knowledge goes under ``projects/<project>/<category>/`` per
wiki-capture; a short id suffix guarantees uniqueness.
"""
from deerflow.knowledge.skill_templates import category_to_dir
cat_dir = category_to_dir(category)
safe_slug = slugify(slug)
suffix = note_id.split("_")[-1][:8]
filename = f"{safe_slug}-{suffix}.md"
if project_name:
return f"projects/{slugify(project_name)}/{cat_dir}/{filename}"
return f"{cat_dir}/{filename}"
# -- scaffold ---------------------------------------------------------
def ensure_scaffold(self, *, project_name: str | None = None) -> None:
"""Create the vault directory tree + base files if missing."""
self.root.mkdir(parents=True, exist_ok=True)
for d in _VAULT_DIRS:
(self.root / d).mkdir(parents=True, exist_ok=True)
if project_name:
base = self.root / "projects" / slugify(project_name)
for d in ("concepts", "references", "synthesis", "journal"):
(base / d).mkdir(parents=True, exist_ok=True)
# Base vault files.
index = self.root / "index.md"
if not index.exists():
index.write_text("# Knowledge Index\n\n_页面索引。每次写入新页面后自动更新。_\n\n", encoding="utf-8")
log = self.root / "log.md"
if not log.exists():
log.write_text("# Activity Log\n\n_CAPTURE / INGEST / QUERY / EXPORT 记录。_\n\n", encoding="utf-8")
hot = self.root / "hot.md"
if not hot.exists():
hot.write_text("# Hot\n\n_近期活动摘要。_\n\n", encoding="utf-8")
taxonomy = self.root / "_meta" / "taxonomy.md"
if not taxonomy.exists():
taxonomy.write_text(
"# Taxonomy\n\n- synthesis — 多步骤分析、方案、结论\n- concepts — 概念、框架、模型\n"
"- references — 外部来源、搜索结果、文章资料\n- decision — 架构或设计决策\n- journal — 完整会话摘要\n",
encoding="utf-8",
)
# -- frontmatter ------------------------------------------------------
@staticmethod
def build_frontmatter(
*,
note_id: str,
title: str,
category: str,
tags: list[str],
sources: list[str],
summary: str,
created: str,
updated: str,
base_confidence: float,
lifecycle: str,
provenance: dict[str, float] | None = None,
relationships: list[dict[str, str]] | None = None,
) -> dict[str, Any]:
"""Build an obsidian-wiki / wiki-capture compatible frontmatter dict."""
created_date = (created or "")[:10]
return {
"id": note_id,
"title": title,
"category": category,
"tags": list(tags or []),
"sources": list(sources or []),
"created": created,
"updated": updated,
"summary": summary or "",
"provenance": provenance or {"extracted": 0.7, "inferred": 0.2, "ambiguous": 0.1},
"base_confidence": round(float(base_confidence or 0.0), 2),
"lifecycle": lifecycle,
"lifecycle_changed": created_date,
"relationships": relationships or [],
}
# -- IO ---------------------------------------------------------------
def write_note(self, rel_path: str, frontmatter: dict[str, Any], body: str) -> Path:
"""Write ``frontmatter`` + ``body`` to ``rel_path`` (creates dirs)."""
path = self.resolve(rel_path)
path.parent.mkdir(parents=True, exist_ok=True)
fm = yaml.safe_dump(frontmatter, allow_unicode=True, sort_keys=False, default_flow_style=False)
path.write_text(f"---\n{fm}---\n\n{body.strip()}\n", encoding="utf-8")
return path
def read_note(self, rel_path: str) -> tuple[dict[str, Any], str] | None:
"""Return ``(frontmatter, body)`` for a note, or ``None`` if missing."""
path = self.resolve(rel_path)
if not path.exists():
return None
return self.parse(path.read_text(encoding="utf-8"))
@staticmethod
def parse(text: str) -> tuple[dict[str, Any], str]:
"""Split raw Markdown into ``(frontmatter dict, body)``."""
match = _FRONTMATTER_RE.match(text or "")
if not match:
return {}, text or ""
try:
fm = yaml.safe_load(match.group(1)) or {}
except yaml.YAMLError:
fm = {}
if not isinstance(fm, dict):
fm = {}
return fm, match.group(2) or ""
def delete_note(self, rel_path: str) -> None:
"""Remove a note file if present (used on hard delete; archive keeps it)."""
path = self.resolve(rel_path)
if path.exists():
path.unlink()
def now_iso_with_tz(dt: datetime) -> str:
"""ISO-8601 string with timezone offset (frontmatter ``created``/``updated``)."""
if dt.tzinfo is None:
return dt.isoformat()
return dt.isoformat(timespec="seconds")