"""obsidian-wiki ``wiki-export`` compatible graph generation. Builds a knowledge graph from the notes (nodes) plus their wikilinks and shared tags (edges) and writes the four native artifacts into ``/wiki-export/``: ``graph.json``, ``graph.graphml``, ``cypher.txt`` and a self-contained ``graph.html`` (no external CDN, safe to iframe offline). """ from __future__ import annotations import html import json from pathlib import Path from typing import Any from deerflow.knowledge.markdown_vault import extract_wikilinks def _vault_key(vault_path: str | None) -> str | None: if not vault_path: return None key = vault_path.replace("\\", "/") if key.endswith(".md"): key = key[:-3] return key def normalize_blacklist(labels: list[str] | set[str] | None) -> set[str]: """Normalize blacklist labels for matching (trimmed, lower-cased, non-empty).""" out: set[str] = set() for raw in labels or []: label = (raw or "").strip().lower() if label: out.add(label) return out def build_graph( notes: list[dict[str, Any]], entities: list[dict[str, Any]] | None = None, relations: list[dict[str, Any]] | None = None, *, blacklist: set[str] | None = None, ) -> dict[str, Any]: """Return ``{nodes, edges}`` from notes (+ optional entities/relations). Note nodes come from ``notes`` (id, title, tags, vault_path, content_md); when ``entities``/``relations`` are supplied (phase 4), entity nodes and note→entity / entity→entity edges are added on top of the wikilink and shared-tag edges. ``blacklist`` is a set of *normalized* labels (see :func:`normalize_blacklist`); notes whose title and entities whose name match are dropped, together with every edge touching them (edges only join surviving nodes). """ blocked = blacklist or set() def _is_blocked(label: str | None) -> bool: return bool(blocked) and (label or "").strip().lower() in blocked # Drop blacklisted notes up front so the wikilink / shared-tag edge loops # below never reference them either. notes = [n for n in notes if not _is_blocked(n.get("title"))] nodes = [] by_key: dict[str, str] = {} note_ids: set[str] = set() for n in notes: nodes.append({"id": n["id"], "type": "note", "label": n.get("title") or n["id"], "category": n.get("source_type")}) note_ids.add(n["id"]) key = _vault_key(n.get("vault_path")) if key: by_key[key] = n["id"] # Only surface entities that are still mentioned by a LIVE note. Entities whose # only referencing notes were deleted/archived would otherwise linger forever # as orphan nodes ("graph doesn't follow my deletions"). mentioned: set[str] = set() for rel in relations or []: if rel.get("from_note_id") in note_ids and rel.get("to_entity_id"): mentioned.add(rel["to_entity_id"]) entity_ids: set[str] = set() for ent in entities or []: if ent["id"] not in mentioned: continue if _is_blocked(ent.get("name")): continue nodes.append({"id": ent["id"], "type": "entity", "label": ent.get("name") or ent["id"], "category": ent.get("entity_type")}) entity_ids.add(ent["id"]) edges: list[dict[str, Any]] = [] edge_seen: set[tuple[str, str, str]] = set() def add_edge(a: str, b: str, label: str, weight: float) -> None: if a == b: return key = (min(a, b), max(a, b), label) if key in edge_seen: return edge_seen.add(key) edges.append({"id": f"rel_{len(edges) + 1}", "source": a, "target": b, "label": label, "weight": weight}) # Wikilink edges. for n in notes: for target in extract_wikilinks(n.get("content_md") or ""): tid = by_key.get(target.replace("\\", "/").removesuffix(".md")) if tid: add_edge(n["id"], tid, "links", 2) # Shared-tag edges. tag_to_notes: dict[str, list[str]] = {} for n in notes: for tag in n.get("tags") or []: tag_to_notes.setdefault(tag.lower(), []).append(n["id"]) for tag, ids in tag_to_notes.items(): if len(ids) < 2 or len(ids) > 50: # skip ubiquitous tags (e.g. knowledge-base) continue for i in range(len(ids)): for j in range(i + 1, len(ids)): add_edge(ids[i], ids[j], "related", 1) # Entity / relation edges (phase 4) — only when both endpoints exist as nodes. valid = note_ids | entity_ids for rel in relations or []: src = rel.get("from_note_id") or rel.get("from_entity_id") tgt = rel.get("to_note_id") or rel.get("to_entity_id") if src in valid and tgt in valid: add_edge(src, tgt, rel.get("relation_type") or "related", float(rel.get("weight") or 1)) return {"nodes": nodes, "edges": edges} def to_graphml(graph: dict[str, Any]) -> str: lines = [ '', '', ' ', ' ', ' ', ] for n in graph["nodes"]: lines.append(f' {html.escape(str(n["label"]))}') for e in graph["edges"]: lines.append( f' {html.escape(str(e["label"]))}' ) lines += [" ", ""] return "\n".join(lines) def to_cypher(graph: dict[str, Any]) -> str: lines = [] for n in graph["nodes"]: label = str(n["label"]).replace("'", "\\'") lines.append(f"CREATE (:Note {{id:'{n['id']}', title:'{label}'}});") for e in graph["edges"]: rel = str(e["label"]).upper() lines.append(f"MATCH (a:Note {{id:'{e['source']}'}}),(b:Note {{id:'{e['target']}'}}) CREATE (a)-[:{rel} {{weight:{e['weight']}}}]->(b);") return "\n".join(lines) def to_html(graph: dict[str, Any]) -> str: """Self-contained, theme-aware force-directed graph (canvas + vanilla JS, no CDN). The page reads ``?theme=light|dark`` from its own URL so the embedding iframe can stay in sync with the app's day/night mode (an iframe cannot inherit the parent's CSS). The layout settles and then **freezes** so a dense graph stops drifting and stays clickable; node size scales with degree; and hovering or clicking a node focuses it + its direct neighbours while dimming everything else, which cuts through heavy node/edge clutter. """ data_json = json.dumps(graph, ensure_ascii=False) return _HTML_TEMPLATE.replace("__GRAPH_DATA__", data_json) def export_all( vault_root: Path | str, notes: list[dict[str, Any]], *, entities: list[dict[str, Any]] | None = None, relations: list[dict[str, Any]] | None = None, write_html: bool = True, write_json: bool = True, blacklist: set[str] | None = None, ) -> dict[str, Any]: """Generate graph artifacts under ``/wiki-export/`` and return stats.""" out_dir = Path(vault_root).resolve() / "wiki-export" out_dir.mkdir(parents=True, exist_ok=True) graph = build_graph(notes, entities, relations, blacklist=blacklist) written: list[str] = [] if write_json: (out_dir / "graph.json").write_text(json.dumps(graph, ensure_ascii=False, indent=2), encoding="utf-8") written.append("graph.json") (out_dir / "graph.graphml").write_text(to_graphml(graph), encoding="utf-8") written.append("graph.graphml") (out_dir / "cypher.txt").write_text(to_cypher(graph), encoding="utf-8") written.append("cypher.txt") if write_html: (out_dir / "graph.html").write_text(to_html(graph), encoding="utf-8") written.append("graph.html") return { "files": written, "stats": {"nodes": len(graph["nodes"]), "edges": len(graph["edges"])}, "graph": graph, } _HTML_TEMPLATE = """ Knowledge Graph
知识笔记
实体
×

重置视图
"""