153 lines
6.2 KiB
Python
153 lines
6.2 KiB
Python
"""IR construction and validation.
|
||
|
||
V1 intentionally starts with deterministic, evidence-backed extraction. The
|
||
schema is compatible with a future model-backed extractor, while never letting
|
||
unvalidated model output flow directly into persistence or graph writers.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import hashlib
|
||
import re
|
||
from typing import Any
|
||
|
||
from deerflow.skill_knowledge.scanner import ScanResult
|
||
|
||
EXTRACTOR_VERSION = "skill-knowledge-v1"
|
||
PROMPT_HASH = f"sha256:{hashlib.sha256(b'skill-knowledge-v1-evidence-only').hexdigest()}"
|
||
PREDICATE_PATTERNS = {
|
||
"depends_on": re.compile(r"(?:依赖|需要|requires?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
|
||
"uses": re.compile(r"(?:调用|使用|uses?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
|
||
"produces": re.compile(r"(?:生成|输出|produces?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
|
||
"consumes": re.compile(r"(?:读取|输入|consumes?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I),
|
||
}
|
||
|
||
|
||
def _evidence(path: str, text: str, start: int, end: int) -> dict[str, Any]:
|
||
line_start = text.count("\n", 0, start) + 1
|
||
line_end = text.count("\n", 0, end) + 1
|
||
quote = " ".join(text[start:end].split())[:240]
|
||
return {"file_path": path, "line_start": line_start, "line_end": line_end, "quote": quote}
|
||
|
||
|
||
def build_skill_ir(scan: ScanResult, *, display_name: str | None = None, description: str = "") -> dict[str, Any]:
|
||
entities: list[dict[str, Any]] = [
|
||
{
|
||
"local_id": "ent_skill",
|
||
"type": "skill",
|
||
"name": display_name or scan.skill_name,
|
||
"aliases": [scan.skill_name] if display_name and display_name != scan.skill_name else [],
|
||
"description": description,
|
||
"attributes": {},
|
||
"evidence": [],
|
||
"confidence": 1.0,
|
||
}
|
||
]
|
||
relations: list[dict[str, Any]] = []
|
||
by_key: dict[tuple[str, str], str] = {("skill", (display_name or scan.skill_name).casefold()): "ent_skill"}
|
||
|
||
for file in scan.files:
|
||
if not file.text:
|
||
continue
|
||
for predicate, pattern in PREDICATE_PATTERNS.items():
|
||
for match in pattern.finditer(file.text):
|
||
name = match.group(1).strip(" `\"'“”()[]")
|
||
if not name or len(name) > 80:
|
||
continue
|
||
key = ("concept", name.casefold())
|
||
entity_id = by_key.get(key)
|
||
if entity_id is None:
|
||
entity_id = f"ent_{len(entities):04d}"
|
||
by_key[key] = entity_id
|
||
entities.append(
|
||
{
|
||
"local_id": entity_id,
|
||
"type": "concept",
|
||
"name": name,
|
||
"aliases": [],
|
||
"description": "从技能业务说明中抽取的知识对象",
|
||
"attributes": {},
|
||
"evidence": [_evidence(file.path, file.text, match.start(), match.end())],
|
||
"confidence": 0.82,
|
||
}
|
||
)
|
||
relations.append(
|
||
{
|
||
"local_id": f"rel_{len(relations):04d}",
|
||
"source_local_id": "ent_skill",
|
||
"predicate": predicate,
|
||
"target_local_id": entity_id,
|
||
"description": " ".join(match.group(0).split())[:240],
|
||
"relation_source": "explicit_text",
|
||
"evidence": [_evidence(file.path, file.text, match.start(), match.end())],
|
||
"confidence": 0.82,
|
||
}
|
||
)
|
||
for ref in file.references:
|
||
name = ref["target_path"]
|
||
key = ("file", name.casefold())
|
||
entity_id = by_key.get(key)
|
||
if entity_id is None:
|
||
entity_id = f"ent_{len(entities):04d}"
|
||
by_key[key] = entity_id
|
||
entities.append(
|
||
{
|
||
"local_id": entity_id,
|
||
"type": "file",
|
||
"name": name,
|
||
"aliases": [],
|
||
"description": "技能文档引用的资料文件",
|
||
"attributes": {},
|
||
"evidence": [{"file_path": file.path, "line_start": 1, "line_end": 1, "quote": name}],
|
||
"confidence": 1.0,
|
||
}
|
||
)
|
||
relations.append(
|
||
{
|
||
"local_id": f"rel_{len(relations):04d}",
|
||
"source_local_id": "ent_skill",
|
||
"predicate": "references",
|
||
"target_local_id": entity_id,
|
||
"description": f"{file.path} 引用 {name}",
|
||
"relation_source": "deterministic_structure",
|
||
"evidence": [{"file_path": file.path, "line_start": 1, "line_end": 1, "quote": name}],
|
||
"confidence": 1.0,
|
||
}
|
||
)
|
||
|
||
title = display_name or scan.skill_name
|
||
summaries = [f"- **{item.title}**(`{item.path}`)" for item in scan.files[:100]]
|
||
content = "\n".join(
|
||
[
|
||
f"# {title}",
|
||
"",
|
||
description or "由技能知识文件安全归纳生成。",
|
||
"",
|
||
"## 知识资料",
|
||
"",
|
||
*(summaries or ["- 暂无可归纳知识文件"]),
|
||
"",
|
||
f"> 源摘要:`{scan.source_digest}`;代码与脚本文件未读取。",
|
||
]
|
||
)
|
||
return {
|
||
"extractor_version": EXTRACTOR_VERSION,
|
||
"prompt_hash": PROMPT_HASH,
|
||
"model_name": "deterministic-v1",
|
||
"skill": {
|
||
"name": scan.skill_name,
|
||
"display_name": title,
|
||
"description": description,
|
||
"source_root": scan.source_root,
|
||
"source_digest": scan.source_digest,
|
||
},
|
||
"files": scan.manifest,
|
||
"entities": entities,
|
||
"relations": relations,
|
||
"wiki": {
|
||
"skill_page": {"slug": f"skills/{scan.skill_name}/overview", "title": title, "content": content},
|
||
"pages": [],
|
||
},
|
||
"safety": {"blocked_files": scan.blocked_files},
|
||
}
|