deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/skill_knowledge/extractor.py
2026-09-07 18:24:55 +08:00

153 lines
6.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""IR construction and validation.
V1 intentionally starts with deterministic, evidence-backed extraction. The
schema is compatible with a future model-backed extractor, while never letting
unvalidated model output flow directly into persistence or graph writers.
"""
from __future__ import annotations
import hashlib
import re
from typing import Any
from deerflow.skill_knowledge.scanner import ScanResult
EXTRACTOR_VERSION = "skill-knowledge-v1"
PROMPT_HASH = f"sha256:{hashlib.sha256(b'skill-knowledge-v1-evidence-only').hexdigest()}"
PREDICATE_PATTERNS = {
"depends_on": re.compile(r"(?:依赖|需要|requires?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
"uses": re.compile(r"(?:调用|使用|uses?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
"produces": re.compile(r"(?:生成|输出|produces?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
"consumes": re.compile(r"(?:读取|输入|consumes?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
}
def _evidence(path: str, text: str, start: int, end: int) -> dict[str, Any]:
line_start = text.count("\n", 0, start) + 1
line_end = text.count("\n", 0, end) + 1
quote = " ".join(text[start:end].split())[:240]
return {"file_path": path, "line_start": line_start, "line_end": line_end, "quote": quote}
def build_skill_ir(scan: ScanResult, *, display_name: str | None = None, description: str = "") -> dict[str, Any]:
entities: list[dict[str, Any]] = [
{
"local_id": "ent_skill",
"type": "skill",
"name": display_name or scan.skill_name,
"aliases": [scan.skill_name] if display_name and display_name != scan.skill_name else [],
"description": description,
"attributes": {},
"evidence": [],
"confidence": 1.0,
}
]
relations: list[dict[str, Any]] = []
by_key: dict[tuple[str, str], str] = {("skill", (display_name or scan.skill_name).casefold()): "ent_skill"}
for file in scan.files:
if not file.text:
continue
for predicate, pattern in PREDICATE_PATTERNS.items():
for match in pattern.finditer(file.text):
name = match.group(1).strip(" `\"'“”()[]")
if not name or len(name) > 80:
continue
key = ("concept", name.casefold())
entity_id = by_key.get(key)
if entity_id is None:
entity_id = f"ent_{len(entities):04d}"
by_key[key] = entity_id
entities.append(
{
"local_id": entity_id,
"type": "concept",
"name": name,
"aliases": [],
"description": "从技能业务说明中抽取的知识对象",
"attributes": {},
"evidence": [_evidence(file.path, file.text, match.start(), match.end())],
"confidence": 0.82,
}
)
relations.append(
{
"local_id": f"rel_{len(relations):04d}",
"source_local_id": "ent_skill",
"predicate": predicate,
"target_local_id": entity_id,
"description": " ".join(match.group(0).split())[:240],
"relation_source": "explicit_text",
"evidence": [_evidence(file.path, file.text, match.start(), match.end())],
"confidence": 0.82,
}
)
for ref in file.references:
name = ref["target_path"]
key = ("file", name.casefold())
entity_id = by_key.get(key)
if entity_id is None:
entity_id = f"ent_{len(entities):04d}"
by_key[key] = entity_id
entities.append(
{
"local_id": entity_id,
"type": "file",
"name": name,
"aliases": [],
"description": "技能文档引用的资料文件",
"attributes": {},
"evidence": [{"file_path": file.path, "line_start": 1, "line_end": 1, "quote": name}],
"confidence": 1.0,
}
)
relations.append(
{
"local_id": f"rel_{len(relations):04d}",
"source_local_id": "ent_skill",
"predicate": "references",
"target_local_id": entity_id,
"description": f"{file.path} 引用 {name}",
"relation_source": "deterministic_structure",
"evidence": [{"file_path": file.path, "line_start": 1, "line_end": 1, "quote": name}],
"confidence": 1.0,
}
)
title = display_name or scan.skill_name
summaries = [f"- **{item.title}**(`{item.path}`)" for item in scan.files[:100]]
content = "\n".join(
[
f"# {title}",
"",
description or "由技能知识文件安全归纳生成。",
"",
"## 知识资料",
"",
*(summaries or ["- 暂无可归纳知识文件"]),
"",
f"> 源摘要:`{scan.source_digest}`;代码与脚本文件未读取。",
]
)
return {
"extractor_version": EXTRACTOR_VERSION,
"prompt_hash": PROMPT_HASH,
"model_name": "deterministic-v1",
"skill": {
"name": scan.skill_name,
"display_name": title,
"description": description,
"source_root": scan.source_root,
"source_digest": scan.source_digest,
},
"files": scan.manifest,
"entities": entities,
"relations": relations,
"wiki": {
"skill_page": {"slug": f"skills/{scan.skill_name}/overview", "title": title, "content": content},
"pages": [],
},
"safety": {"blocked_files": scan.blocked_files},
}