deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/skill_knowledge/scanner.py
2026-09-07 18:24:55 +08:00

352 lines
12 KiB
Python

"""Safe, deterministic scanner for the knowledge-bearing part of a skill."""
from __future__ import annotations
import csv
import hashlib
import json
import re
import zipfile
from dataclasses import dataclass, field
from pathlib import Path
from typing import Any
CODE_SUFFIXES = {
".py",
".pyc",
".pyo",
".ts",
".tsx",
".js",
".jsx",
".mjs",
".cjs",
".sh",
".bash",
".zsh",
".fish",
".bat",
".cmd",
".ps1",
".psm1",
".java",
".kt",
".kts",
".go",
".rs",
".c",
".cc",
".cpp",
".h",
".hpp",
".cs",
".php",
".rb",
".pl",
".lua",
".swift",
".scala",
".sql",
".vue",
".svelte",
}
TEXT_SUFFIXES = {".md", ".mdx", ".txt", ".json", ".yaml", ".yml", ".toml", ".ini", ".csv", ".tsv"}
BINARY_KNOWLEDGE_SUFFIXES = {".xlsx", ".xls", ".pdf", ".docx", ".pptx", ".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".zip"}
EXCLUDED_DIRS = {".git", "node_modules", "__pycache__", ".pytest_cache", "dist", "build", ".mypy_cache", ".ruff_cache", ".address-edits"}
EXCLUDED_FILENAMES = {
"package-lock.json",
"pnpm-lock.yaml",
"yarn.lock",
"bun.lockb",
"poetry.lock",
"uv.lock",
"pipfile.lock",
"cargo.lock",
"composer.lock",
}
SENSITIVE_NAME_RE = re.compile(
r"(^|[._-])(\.env|id_rsa|id_dsa|id_ecdsa|id_ed25519|private[_-]?key|credentials?|secrets?)([._-]|$)",
re.IGNORECASE,
)
SECRET_CONTENT_RE = re.compile(r"(?im)^\s*(?:api[_-]?key|access[_-]?key|secret(?:[_-]?key)?|password|token|private[_-]?key)\s*[:=]\s*[^\s]{8,}\s*$")
PEM_PRIVATE_RE = re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----")
FENCE_RE = re.compile(r"(?ms)^\s*(```|~~~).*?^\s*\1\s*$")
MARKDOWN_LINK_RE = re.compile(r"\[[^\]]+\]\(([^)#?]+)(?:#[^)]*)?\)")
MAX_TEXT_BYTES = 2 * 1024 * 1024
MAX_ARCHIVE_FILES = 1_000
MAX_ARCHIVE_ENTRY_BYTES = 20 * 1024 * 1024
MAX_ARCHIVE_TOTAL_BYTES = 200 * 1024 * 1024
@dataclass(slots=True)
class ScannedFile:
path: str
kind: str
sha256: str
size: int
text: str = ""
title: str = ""
references: list[dict[str, Any]] = field(default_factory=list)
safety: dict[str, Any] = field(default_factory=lambda: {"status": "allowed", "warnings": []})
def manifest(self) -> dict[str, Any]:
return {
"path": self.path,
"kind": self.kind,
"sha256": self.sha256,
"size": self.size,
"text_digest": f"sha256:{hashlib.sha256(self.text.encode('utf-8')).hexdigest()}" if self.text else None,
"title": self.title,
"references": self.references,
"safety": self.safety,
}
@dataclass(slots=True)
class ScanResult:
skill_name: str
source_root: str
source_digest: str
files: list[ScannedFile]
blocked_files: list[dict[str, str]]
skipped_code_count: int
skipped_other_count: int
@property
def manifest(self) -> list[dict[str, Any]]:
return [item.manifest() for item in self.files] + [{"path": item["path"], "kind": "blocked", "safety": {"status": "blocked", "warnings": [item["reason"]]}} for item in self.blocked_files]
@property
def counts(self) -> dict[str, int]:
return {
"knowledge_files": len(self.files),
"blocked_files": len(self.blocked_files),
"skipped_code_files": self.skipped_code_count,
"skipped_other_files": self.skipped_other_count,
}
def _safe_relative(path: Path, root: Path) -> str:
resolved = path.resolve()
resolved_root = root.resolve()
try:
relative = resolved.relative_to(resolved_root)
except ValueError as exc:
raise ValueError("Skill file escapes its source root") from exc
return relative.as_posix()
def _strip_markdown_code(text: str) -> str:
return FENCE_RE.sub("\n[示例代码已按安全策略省略]\n", text)
def _read_text(path: Path, suffix: str) -> str:
raw = path.read_bytes()
if len(raw) > MAX_TEXT_BYTES:
raise ValueError("text file exceeds 2 MiB parsing limit")
text = raw.decode("utf-8-sig", errors="replace")
if suffix in {".md", ".mdx"}:
return _strip_markdown_code(text)
if suffix in {".csv", ".tsv"}:
delimiter = "\t" if suffix == ".tsv" else ","
rows = list(csv.reader(text.splitlines(), delimiter=delimiter))
return "\n".join(" | ".join(cell.strip() for cell in row) for row in rows[:500])
if suffix == ".json":
try:
return json.dumps(json.loads(text), ensure_ascii=False, indent=2)
except json.JSONDecodeError:
return text
return text
def _title(text: str, fallback: str) -> str:
match = re.search(r"(?m)^#\s+(.+?)\s*$", text)
return match.group(1).strip() if match else fallback
def _kind(suffix: str) -> str:
return {
".md": "markdown",
".mdx": "markdown",
".txt": "text",
".csv": "table",
".tsv": "table",
".xlsx": "spreadsheet",
".xls": "spreadsheet",
".pdf": "pdf",
".docx": "office",
".pptx": "office",
".png": "image",
".jpg": "image",
".jpeg": "image",
".webp": "image",
".gif": "image",
".bmp": "image",
".zip": "archive",
}.get(suffix, "structured_data")
def _zip_members(path: Path, archive_relative: str) -> tuple[list[ScannedFile], list[dict[str, str]], int, int]:
files: list[ScannedFile] = []
blocked: list[dict[str, str]] = []
skipped_code = 0
skipped_other = 0
with zipfile.ZipFile(path) as archive:
members = archive.infolist()
if len(members) > MAX_ARCHIVE_FILES:
return [], [{"path": archive_relative, "reason": "压缩包文件数超过安全上限"}], 0, 0
total_size = sum(member.file_size for member in members)
if total_size > MAX_ARCHIVE_TOTAL_BYTES:
return [], [{"path": archive_relative, "reason": "压缩包展开体积超过安全上限"}], 0, 0
for member in members:
member_path = Path(member.filename.replace("\\", "/"))
virtual_path = f"{archive_relative}!/{member_path.as_posix()}"
unix_mode = member.external_attr >> 16
if member.is_dir():
continue
if member_path.is_absolute() or ".." in member_path.parts or (unix_mode & 0o170000) == 0o120000:
blocked.append({"path": virtual_path, "reason": "压缩包包含路径穿越或符号链接"})
continue
suffix = member_path.suffix.lower()
lower_name = member_path.name.lower()
if lower_name in EXCLUDED_FILENAMES or suffix in CODE_SUFFIXES:
skipped_code += 1
continue
if SENSITIVE_NAME_RE.search(lower_name) or suffix in {".pem", ".key", ".p12", ".pfx"}:
blocked.append({"path": virtual_path, "reason": "疑似密钥或敏感配置文件"})
continue
if suffix not in TEXT_SUFFIXES or member.file_size > MAX_ARCHIVE_ENTRY_BYTES:
skipped_other += 1
continue
raw = archive.read(member)
text = raw.decode("utf-8-sig", errors="replace")
if suffix in {".md", ".mdx"}:
text = _strip_markdown_code(text)
if SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text):
blocked.append({"path": virtual_path, "reason": "内容命中高置信敏感信息规则"})
continue
files.append(
ScannedFile(
path=virtual_path,
kind=_kind(suffix),
sha256=hashlib.sha256(raw).hexdigest(),
size=len(raw),
text=text,
title=_title(text, member_path.stem),
)
)
return files, blocked, skipped_code, skipped_other
def scan_skill(skill_name: str, skill_root: Path) -> ScanResult:
"""Scan one already-resolved skill directory without following symlinks."""
root = skill_root.resolve()
if not root.is_dir():
raise FileNotFoundError(f"Skill '{skill_name}' source directory does not exist")
files: list[ScannedFile] = []
blocked: list[dict[str, str]] = []
skipped_code = 0
skipped_other = 0
for path in sorted(root.rglob("*"), key=lambda item: item.as_posix().lower()):
if path.is_symlink() or not path.is_file():
continue
relative = _safe_relative(path, root)
if any(part.lower() in EXCLUDED_DIRS for part in Path(relative).parts[:-1]):
continue
lower_name = path.name.lower()
suffix = path.suffix.lower()
if lower_name in EXCLUDED_FILENAMES or suffix in CODE_SUFFIXES:
skipped_code += 1
continue
if SENSITIVE_NAME_RE.search(lower_name) or suffix in {".pem", ".key", ".p12", ".pfx"}:
blocked.append({"path": relative, "reason": "疑似密钥或敏感配置文件"})
continue
if suffix not in TEXT_SUFFIXES | BINARY_KNOWLEDGE_SUFFIXES:
skipped_other += 1
continue
# Never upload an archive wholesale: it may contain excluded code or a
# blocked secret. Only independently validated knowledge members enter
# the manifest and digest.
if suffix == ".zip":
try:
nested, nested_blocked, nested_code, nested_other = _zip_members(path, relative)
files.extend(nested)
blocked.extend(nested_blocked)
skipped_code += nested_code
skipped_other += nested_other
except (OSError, zipfile.BadZipFile):
blocked.append({"path": relative, "reason": "压缩包损坏或格式不受支持"})
continue
raw = path.read_bytes()
digest = hashlib.sha256(raw).hexdigest()
text = ""
warnings: list[str] = []
if suffix in TEXT_SUFFIXES:
try:
text = _read_text(path, suffix)
except ValueError as exc:
warnings.append(str(exc))
if text and (SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text)):
blocked.append({"path": relative, "reason": "内容命中高置信敏感信息规则"})
continue
references = []
if suffix in {".md", ".mdx"}:
for target in MARKDOWN_LINK_RE.findall(text):
normalized = target.replace("\\", "/").lstrip("./")
if normalized:
references.append({"target_path": normalized, "evidence": "markdown_link", "confidence": 1.0})
files.append(
ScannedFile(
path=relative,
kind=_kind(suffix),
sha256=digest,
size=len(raw),
text=text,
title=_title(text, path.stem),
references=references,
safety={"status": "allowed", "warnings": warnings},
)
)
digest_input = "\n".join(f"{item.path}\0{item.sha256}" for item in files).encode("utf-8")
source_digest = f"sha256:{hashlib.sha256(digest_input).hexdigest()}"
return ScanResult(
skill_name=skill_name,
source_root=str(root),
source_digest=source_digest,
files=files,
blocked_files=blocked,
skipped_code_count=skipped_code,
skipped_other_count=skipped_other,
)
async def enrich_converted_documents(scan: ScanResult) -> ScanResult:
"""Extract Office/PDF/spreadsheet text without modifying the skill tree."""
from deerflow.utils.file_conversion import CONVERTIBLE_EXTENSIONS, extract_file_to_markdown
allowed: list[ScannedFile] = []
root = Path(scan.source_root)
for item in scan.files:
if Path(item.path).suffix.lower() in CONVERTIBLE_EXTENSIONS:
text = await extract_file_to_markdown(root / item.path)
if text:
text = _strip_markdown_code(text)
if SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text):
scan.blocked_files.append({"path": item.path, "reason": "转换文本命中高置信敏感信息规则"})
continue
item.text = text
item.title = _title(text, item.title)
else:
item.safety["warnings"].append("文档转换不可用,仅保留原始附件")
allowed.append(item)
scan.files = allowed
digest_input = "\n".join(f"{item.path}\0{item.sha256}" for item in scan.files).encode("utf-8")
scan.source_digest = f"sha256:{hashlib.sha256(digest_input).hexdigest()}"
return scan