"""IR construction and validation. V1 intentionally starts with deterministic, evidence-backed extraction. The schema is compatible with a future model-backed extractor, while never letting unvalidated model output flow directly into persistence or graph writers. """ from __future__ import annotations import hashlib import re from typing import Any from deerflow.skill_knowledge.scanner import ScanResult EXTRACTOR_VERSION = "skill-knowledge-v1" PROMPT_HASH = f"sha256:{hashlib.sha256(b'skill-knowledge-v1-evidence-only').hexdigest()}" PREDICATE_PATTERNS = { "depends_on": re.compile(r"(?:依赖|需要|requires?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I), "uses": re.compile(r"(?:调用|使用|uses?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I), "produces": re.compile(r"(?:生成|输出|produces?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I), "consumes": re.compile(r"(?:读取|输入|consumes?)\s*[`\"“]?([^,。;;\n`\"”]{2,80})", re.I), } def _evidence(path: str, text: str, start: int, end: int) -> dict[str, Any]: line_start = text.count("\n", 0, start) + 1 line_end = text.count("\n", 0, end) + 1 quote = " ".join(text[start:end].split())[:240] return {"file_path": path, "line_start": line_start, "line_end": line_end, "quote": quote} def build_skill_ir(scan: ScanResult, *, display_name: str | None = None, description: str = "") -> dict[str, Any]: entities: list[dict[str, Any]] = [ { "local_id": "ent_skill", "type": "skill", "name": display_name or scan.skill_name, "aliases": [scan.skill_name] if display_name and display_name != scan.skill_name else [], "description": description, "attributes": {}, "evidence": [], "confidence": 1.0, } ] relations: list[dict[str, Any]] = [] by_key: dict[tuple[str, str], str] = {("skill", (display_name or scan.skill_name).casefold()): "ent_skill"} for file in scan.files: if not file.text: continue for predicate, pattern in PREDICATE_PATTERNS.items(): for match in pattern.finditer(file.text): name = match.group(1).strip(" `\"'“”()[]") if not name or len(name) > 80: continue key = ("concept", name.casefold()) entity_id = by_key.get(key) if entity_id is None: entity_id = f"ent_{len(entities):04d}" by_key[key] = entity_id entities.append( { "local_id": entity_id, "type": "concept", "name": name, "aliases": [], "description": "从技能业务说明中抽取的知识对象", "attributes": {}, "evidence": [_evidence(file.path, file.text, match.start(), match.end())], "confidence": 0.82, } ) relations.append( { "local_id": f"rel_{len(relations):04d}", "source_local_id": "ent_skill", "predicate": predicate, "target_local_id": entity_id, "description": " ".join(match.group(0).split())[:240], "relation_source": "explicit_text", "evidence": [_evidence(file.path, file.text, match.start(), match.end())], "confidence": 0.82, } ) for ref in file.references: name = ref["target_path"] key = ("file", name.casefold()) entity_id = by_key.get(key) if entity_id is None: entity_id = f"ent_{len(entities):04d}" by_key[key] = entity_id entities.append( { "local_id": entity_id, "type": "file", "name": name, "aliases": [], "description": "技能文档引用的资料文件", "attributes": {}, "evidence": [{"file_path": file.path, "line_start": 1, "line_end": 1, "quote": name}], "confidence": 1.0, } ) relations.append( { "local_id": f"rel_{len(relations):04d}", "source_local_id": "ent_skill", "predicate": "references", "target_local_id": entity_id, "description": f"{file.path} 引用 {name}", "relation_source": "deterministic_structure", "evidence": [{"file_path": file.path, "line_start": 1, "line_end": 1, "quote": name}], "confidence": 1.0, } ) title = display_name or scan.skill_name summaries = [f"- **{item.title}**(`{item.path}`)" for item in scan.files[:100]] content = "\n".join( [ f"# {title}", "", description or "由技能知识文件安全归纳生成。", "", "## 知识资料", "", *(summaries or ["- 暂无可归纳知识文件"]), "", f"> 源摘要:`{scan.source_digest}`;代码与脚本文件未读取。", ] ) return { "extractor_version": EXTRACTOR_VERSION, "prompt_hash": PROMPT_HASH, "model_name": "deterministic-v1", "skill": { "name": scan.skill_name, "display_name": title, "description": description, "source_root": scan.source_root, "source_digest": scan.source_digest, }, "files": scan.manifest, "entities": entities, "relations": relations, "wiki": { "skill_page": {"slug": f"skills/{scan.skill_name}/overview", "title": title, "content": content}, "pages": [], }, "safety": {"blocked_files": scan.blocked_files}, }