deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/config/knowledge_config.py
2026-09-07 18:24:55 +08:00

76 lines
5.4 KiB
Python

"""Configuration for the product knowledge base (obsidian-wiki compatible)."""
from pydantic import BaseModel, Field
class KnowledgeExportConfig(BaseModel):
"""Knowledge graph export switches (``wiki-export`` compatible output)."""
enabled: bool = Field(default=True, description="Master switch for graph export.")
graph_html_enabled: bool = Field(default=True, description="Generate wiki-export/graph.html.")
graph_json_enabled: bool = Field(default=True, description="Generate wiki-export/graph.json.")
class KnowledgeQueryConfig(BaseModel):
"""Layered query strategy flags (``wiki-query`` style)."""
use_hot_md: bool = Field(default=True, description="Consult hot.md when searching.")
use_index_md: bool = Field(default=True, description="Consult index.md when searching.")
use_frontmatter_scan: bool = Field(default=True, description="Scan frontmatter when searching.")
class KnowledgeEmbeddingConfig(BaseModel):
"""Vector embedding config (phase 2).
Vector RAG is opt-in: set ``enabled: true`` and point ``base_url`` at an
OpenAI-compatible ``/embeddings`` endpoint. When disabled or unreachable,
search transparently falls back to keyword scoring, so the feature degrades
gracefully on offline deployments without an embedding model.
"""
enabled: bool = Field(default=False, description="Enable vector embeddings.")
provider: str = Field(default="openai", description="Embedding provider (openai-compatible).")
model: str = Field(default="", description="Embedding model name, e.g. text-embedding-3-small / bge-m3.")
base_url: str = Field(default="", description="OpenAI-compatible base URL (…/v1). Empty disables vector mode.")
api_key: str = Field(default="", description="API key for the embedding endpoint ($ENV resolved like other config).")
dimensions: int | None = Field(default=None, description="Optional embedding dimensions hint sent to the endpoint.")
batch_size: int = Field(default=32, description="Texts per embedding request.")
timeout_seconds: int = Field(default=30, description="HTTP timeout per embedding request.")
class KnowledgeConfig(BaseModel):
"""Product knowledge base configuration.
The first release ships a global, shared, obsidian-wiki compatible vault.
Knowledge metadata is mirrored into the database for fast listing/search,
while ``content_md`` is written to a Markdown vault that stays compatible
with ``ar9av/obsidian-wiki`` (index.md / log.md / hot.md / .manifest.json,
frontmatter, wikilinks, wiki-export graph artifacts).
"""
enabled: bool = Field(default=True, description="Enable the knowledge base feature.")
provider: str = Field(default="obsidian_wiki", description="Knowledge organization provider.")
vault_dir: str = Field(
default="data/knowledge/obsidian-vault",
description="Vault directory (relative to DEER_FLOW_HOME, or absolute).",
)
project_name: str = Field(default="zncm", description="Project namespace for projects/<name>/ pages.")
auto_ingest_enabled: bool = Field(default=False, description="Auto-sediment conversations in the background (phase 3).")
auto_ingest_min_chars: int = Field(default=300, description="Minimum characters before auto-ingest considers a turn.")
llm_extract_enabled: bool = Field(default=True, description="Distill notes with an LLM (wiki-capture style) instead of the rule-based fallback.")
extract_model_name: str | None = Field(default=None, description="Model (from models[]) used to distill knowledge. Falls back to models[0] when unset.")
extract_max_input_chars: int = Field(default=12000, description="Max characters of conversation transcript fed to the extraction LLM.")
max_thread_messages: int = Field(default=80, description="Maximum thread messages to scan when capturing.")
max_source_items: int = Field(default=20, description="Maximum source items recorded per note.")
related_notes_enabled: bool = Field(default=True, description="On capture, find existing notes on the same topic and link them under ## Related (real cross-references instead of placeholders).")
related_notes_limit: int = Field(default=3, description="Maximum real related notes linked into each captured note.")
multi_note_enabled: bool = Field(default=True, description="Allow the LLM extractor to split one conversation into multiple cross-linked knowledge notes.")
rag_enabled: bool = Field(default=False, description="Enable RAG injection before agent runs (phase 2).")
rag_limit: int = Field(default=5, description="Maximum knowledge snippets injected per query.")
sensitive_redaction_enabled: bool = Field(default=True, description="Redact secrets/PII (tokens, keys, phone, id-card, email) before persisting notes (phase 3).")
dedup_similarity_threshold: float = Field(default=0.82, description="Title/summary similarity above which two notes are flagged as duplicates (phase 3).")
entity_extraction_enabled: bool = Field(default=True, description="Extract entities/relations during LLM distillation and feed them into the graph (phase 4).")
export: KnowledgeExportConfig = Field(default_factory=KnowledgeExportConfig, description="Graph export configuration.")
query: KnowledgeQueryConfig = Field(default_factory=KnowledgeQueryConfig, description="Layered query configuration.")
embedding: KnowledgeEmbeddingConfig = Field(default_factory=KnowledgeEmbeddingConfig, description="Embedding configuration.")