352 lines
12 KiB
Python
352 lines
12 KiB
Python
"""Safe, deterministic scanner for the knowledge-bearing part of a skill."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import hashlib
|
|
import json
|
|
import re
|
|
import zipfile
|
|
from dataclasses import dataclass, field
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
CODE_SUFFIXES = {
|
|
".py",
|
|
".pyc",
|
|
".pyo",
|
|
".ts",
|
|
".tsx",
|
|
".js",
|
|
".jsx",
|
|
".mjs",
|
|
".cjs",
|
|
".sh",
|
|
".bash",
|
|
".zsh",
|
|
".fish",
|
|
".bat",
|
|
".cmd",
|
|
".ps1",
|
|
".psm1",
|
|
".java",
|
|
".kt",
|
|
".kts",
|
|
".go",
|
|
".rs",
|
|
".c",
|
|
".cc",
|
|
".cpp",
|
|
".h",
|
|
".hpp",
|
|
".cs",
|
|
".php",
|
|
".rb",
|
|
".pl",
|
|
".lua",
|
|
".swift",
|
|
".scala",
|
|
".sql",
|
|
".vue",
|
|
".svelte",
|
|
}
|
|
TEXT_SUFFIXES = {".md", ".mdx", ".txt", ".json", ".yaml", ".yml", ".toml", ".ini", ".csv", ".tsv"}
|
|
BINARY_KNOWLEDGE_SUFFIXES = {".xlsx", ".xls", ".pdf", ".docx", ".pptx", ".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".zip"}
|
|
EXCLUDED_DIRS = {".git", "node_modules", "__pycache__", ".pytest_cache", "dist", "build", ".mypy_cache", ".ruff_cache", ".address-edits"}
|
|
EXCLUDED_FILENAMES = {
|
|
"package-lock.json",
|
|
"pnpm-lock.yaml",
|
|
"yarn.lock",
|
|
"bun.lockb",
|
|
"poetry.lock",
|
|
"uv.lock",
|
|
"pipfile.lock",
|
|
"cargo.lock",
|
|
"composer.lock",
|
|
}
|
|
SENSITIVE_NAME_RE = re.compile(
|
|
r"(^|[._-])(\.env|id_rsa|id_dsa|id_ecdsa|id_ed25519|private[_-]?key|credentials?|secrets?)([._-]|$)",
|
|
re.IGNORECASE,
|
|
)
|
|
SECRET_CONTENT_RE = re.compile(r"(?im)^\s*(?:api[_-]?key|access[_-]?key|secret(?:[_-]?key)?|password|token|private[_-]?key)\s*[:=]\s*[^\s]{8,}\s*$")
|
|
PEM_PRIVATE_RE = re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----")
|
|
FENCE_RE = re.compile(r"(?ms)^\s*(```|~~~).*?^\s*\1\s*$")
|
|
MARKDOWN_LINK_RE = re.compile(r"\[[^\]]+\]\(([^)#?]+)(?:#[^)]*)?\)")
|
|
MAX_TEXT_BYTES = 2 * 1024 * 1024
|
|
MAX_ARCHIVE_FILES = 1_000
|
|
MAX_ARCHIVE_ENTRY_BYTES = 20 * 1024 * 1024
|
|
MAX_ARCHIVE_TOTAL_BYTES = 200 * 1024 * 1024
|
|
|
|
|
|
@dataclass(slots=True)
|
|
class ScannedFile:
|
|
path: str
|
|
kind: str
|
|
sha256: str
|
|
size: int
|
|
text: str = ""
|
|
title: str = ""
|
|
references: list[dict[str, Any]] = field(default_factory=list)
|
|
safety: dict[str, Any] = field(default_factory=lambda: {"status": "allowed", "warnings": []})
|
|
|
|
def manifest(self) -> dict[str, Any]:
|
|
return {
|
|
"path": self.path,
|
|
"kind": self.kind,
|
|
"sha256": self.sha256,
|
|
"size": self.size,
|
|
"text_digest": f"sha256:{hashlib.sha256(self.text.encode('utf-8')).hexdigest()}" if self.text else None,
|
|
"title": self.title,
|
|
"references": self.references,
|
|
"safety": self.safety,
|
|
}
|
|
|
|
|
|
@dataclass(slots=True)
|
|
class ScanResult:
|
|
skill_name: str
|
|
source_root: str
|
|
source_digest: str
|
|
files: list[ScannedFile]
|
|
blocked_files: list[dict[str, str]]
|
|
skipped_code_count: int
|
|
skipped_other_count: int
|
|
|
|
@property
|
|
def manifest(self) -> list[dict[str, Any]]:
|
|
return [item.manifest() for item in self.files] + [{"path": item["path"], "kind": "blocked", "safety": {"status": "blocked", "warnings": [item["reason"]]}} for item in self.blocked_files]
|
|
|
|
@property
|
|
def counts(self) -> dict[str, int]:
|
|
return {
|
|
"knowledge_files": len(self.files),
|
|
"blocked_files": len(self.blocked_files),
|
|
"skipped_code_files": self.skipped_code_count,
|
|
"skipped_other_files": self.skipped_other_count,
|
|
}
|
|
|
|
|
|
def _safe_relative(path: Path, root: Path) -> str:
|
|
resolved = path.resolve()
|
|
resolved_root = root.resolve()
|
|
try:
|
|
relative = resolved.relative_to(resolved_root)
|
|
except ValueError as exc:
|
|
raise ValueError("Skill file escapes its source root") from exc
|
|
return relative.as_posix()
|
|
|
|
|
|
def _strip_markdown_code(text: str) -> str:
|
|
return FENCE_RE.sub("\n[示例代码已按安全策略省略]\n", text)
|
|
|
|
|
|
def _read_text(path: Path, suffix: str) -> str:
|
|
raw = path.read_bytes()
|
|
if len(raw) > MAX_TEXT_BYTES:
|
|
raise ValueError("text file exceeds 2 MiB parsing limit")
|
|
text = raw.decode("utf-8-sig", errors="replace")
|
|
if suffix in {".md", ".mdx"}:
|
|
return _strip_markdown_code(text)
|
|
if suffix in {".csv", ".tsv"}:
|
|
delimiter = "\t" if suffix == ".tsv" else ","
|
|
rows = list(csv.reader(text.splitlines(), delimiter=delimiter))
|
|
return "\n".join(" | ".join(cell.strip() for cell in row) for row in rows[:500])
|
|
if suffix == ".json":
|
|
try:
|
|
return json.dumps(json.loads(text), ensure_ascii=False, indent=2)
|
|
except json.JSONDecodeError:
|
|
return text
|
|
return text
|
|
|
|
|
|
def _title(text: str, fallback: str) -> str:
|
|
match = re.search(r"(?m)^#\s+(.+?)\s*$", text)
|
|
return match.group(1).strip() if match else fallback
|
|
|
|
|
|
def _kind(suffix: str) -> str:
|
|
return {
|
|
".md": "markdown",
|
|
".mdx": "markdown",
|
|
".txt": "text",
|
|
".csv": "table",
|
|
".tsv": "table",
|
|
".xlsx": "spreadsheet",
|
|
".xls": "spreadsheet",
|
|
".pdf": "pdf",
|
|
".docx": "office",
|
|
".pptx": "office",
|
|
".png": "image",
|
|
".jpg": "image",
|
|
".jpeg": "image",
|
|
".webp": "image",
|
|
".gif": "image",
|
|
".bmp": "image",
|
|
".zip": "archive",
|
|
}.get(suffix, "structured_data")
|
|
|
|
|
|
def _zip_members(path: Path, archive_relative: str) -> tuple[list[ScannedFile], list[dict[str, str]], int, int]:
|
|
files: list[ScannedFile] = []
|
|
blocked: list[dict[str, str]] = []
|
|
skipped_code = 0
|
|
skipped_other = 0
|
|
with zipfile.ZipFile(path) as archive:
|
|
members = archive.infolist()
|
|
if len(members) > MAX_ARCHIVE_FILES:
|
|
return [], [{"path": archive_relative, "reason": "压缩包文件数超过安全上限"}], 0, 0
|
|
total_size = sum(member.file_size for member in members)
|
|
if total_size > MAX_ARCHIVE_TOTAL_BYTES:
|
|
return [], [{"path": archive_relative, "reason": "压缩包展开体积超过安全上限"}], 0, 0
|
|
for member in members:
|
|
member_path = Path(member.filename.replace("\\", "/"))
|
|
virtual_path = f"{archive_relative}!/{member_path.as_posix()}"
|
|
unix_mode = member.external_attr >> 16
|
|
if member.is_dir():
|
|
continue
|
|
if member_path.is_absolute() or ".." in member_path.parts or (unix_mode & 0o170000) == 0o120000:
|
|
blocked.append({"path": virtual_path, "reason": "压缩包包含路径穿越或符号链接"})
|
|
continue
|
|
suffix = member_path.suffix.lower()
|
|
lower_name = member_path.name.lower()
|
|
if lower_name in EXCLUDED_FILENAMES or suffix in CODE_SUFFIXES:
|
|
skipped_code += 1
|
|
continue
|
|
if SENSITIVE_NAME_RE.search(lower_name) or suffix in {".pem", ".key", ".p12", ".pfx"}:
|
|
blocked.append({"path": virtual_path, "reason": "疑似密钥或敏感配置文件"})
|
|
continue
|
|
if suffix not in TEXT_SUFFIXES or member.file_size > MAX_ARCHIVE_ENTRY_BYTES:
|
|
skipped_other += 1
|
|
continue
|
|
raw = archive.read(member)
|
|
text = raw.decode("utf-8-sig", errors="replace")
|
|
if suffix in {".md", ".mdx"}:
|
|
text = _strip_markdown_code(text)
|
|
if SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text):
|
|
blocked.append({"path": virtual_path, "reason": "内容命中高置信敏感信息规则"})
|
|
continue
|
|
files.append(
|
|
ScannedFile(
|
|
path=virtual_path,
|
|
kind=_kind(suffix),
|
|
sha256=hashlib.sha256(raw).hexdigest(),
|
|
size=len(raw),
|
|
text=text,
|
|
title=_title(text, member_path.stem),
|
|
)
|
|
)
|
|
return files, blocked, skipped_code, skipped_other
|
|
|
|
|
|
def scan_skill(skill_name: str, skill_root: Path) -> ScanResult:
|
|
"""Scan one already-resolved skill directory without following symlinks."""
|
|
|
|
root = skill_root.resolve()
|
|
if not root.is_dir():
|
|
raise FileNotFoundError(f"Skill '{skill_name}' source directory does not exist")
|
|
files: list[ScannedFile] = []
|
|
blocked: list[dict[str, str]] = []
|
|
skipped_code = 0
|
|
skipped_other = 0
|
|
|
|
for path in sorted(root.rglob("*"), key=lambda item: item.as_posix().lower()):
|
|
if path.is_symlink() or not path.is_file():
|
|
continue
|
|
relative = _safe_relative(path, root)
|
|
if any(part.lower() in EXCLUDED_DIRS for part in Path(relative).parts[:-1]):
|
|
continue
|
|
lower_name = path.name.lower()
|
|
suffix = path.suffix.lower()
|
|
if lower_name in EXCLUDED_FILENAMES or suffix in CODE_SUFFIXES:
|
|
skipped_code += 1
|
|
continue
|
|
if SENSITIVE_NAME_RE.search(lower_name) or suffix in {".pem", ".key", ".p12", ".pfx"}:
|
|
blocked.append({"path": relative, "reason": "疑似密钥或敏感配置文件"})
|
|
continue
|
|
if suffix not in TEXT_SUFFIXES | BINARY_KNOWLEDGE_SUFFIXES:
|
|
skipped_other += 1
|
|
continue
|
|
|
|
# Never upload an archive wholesale: it may contain excluded code or a
|
|
# blocked secret. Only independently validated knowledge members enter
|
|
# the manifest and digest.
|
|
if suffix == ".zip":
|
|
try:
|
|
nested, nested_blocked, nested_code, nested_other = _zip_members(path, relative)
|
|
files.extend(nested)
|
|
blocked.extend(nested_blocked)
|
|
skipped_code += nested_code
|
|
skipped_other += nested_other
|
|
except (OSError, zipfile.BadZipFile):
|
|
blocked.append({"path": relative, "reason": "压缩包损坏或格式不受支持"})
|
|
continue
|
|
|
|
raw = path.read_bytes()
|
|
digest = hashlib.sha256(raw).hexdigest()
|
|
text = ""
|
|
warnings: list[str] = []
|
|
if suffix in TEXT_SUFFIXES:
|
|
try:
|
|
text = _read_text(path, suffix)
|
|
except ValueError as exc:
|
|
warnings.append(str(exc))
|
|
if text and (SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text)):
|
|
blocked.append({"path": relative, "reason": "内容命中高置信敏感信息规则"})
|
|
continue
|
|
references = []
|
|
if suffix in {".md", ".mdx"}:
|
|
for target in MARKDOWN_LINK_RE.findall(text):
|
|
normalized = target.replace("\\", "/").lstrip("./")
|
|
if normalized:
|
|
references.append({"target_path": normalized, "evidence": "markdown_link", "confidence": 1.0})
|
|
files.append(
|
|
ScannedFile(
|
|
path=relative,
|
|
kind=_kind(suffix),
|
|
sha256=digest,
|
|
size=len(raw),
|
|
text=text,
|
|
title=_title(text, path.stem),
|
|
references=references,
|
|
safety={"status": "allowed", "warnings": warnings},
|
|
)
|
|
)
|
|
|
|
digest_input = "\n".join(f"{item.path}\0{item.sha256}" for item in files).encode("utf-8")
|
|
source_digest = f"sha256:{hashlib.sha256(digest_input).hexdigest()}"
|
|
return ScanResult(
|
|
skill_name=skill_name,
|
|
source_root=str(root),
|
|
source_digest=source_digest,
|
|
files=files,
|
|
blocked_files=blocked,
|
|
skipped_code_count=skipped_code,
|
|
skipped_other_count=skipped_other,
|
|
)
|
|
|
|
|
|
async def enrich_converted_documents(scan: ScanResult) -> ScanResult:
|
|
"""Extract Office/PDF/spreadsheet text without modifying the skill tree."""
|
|
|
|
from deerflow.utils.file_conversion import CONVERTIBLE_EXTENSIONS, extract_file_to_markdown
|
|
|
|
allowed: list[ScannedFile] = []
|
|
root = Path(scan.source_root)
|
|
for item in scan.files:
|
|
if Path(item.path).suffix.lower() in CONVERTIBLE_EXTENSIONS:
|
|
text = await extract_file_to_markdown(root / item.path)
|
|
if text:
|
|
text = _strip_markdown_code(text)
|
|
if SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text):
|
|
scan.blocked_files.append({"path": item.path, "reason": "转换文本命中高置信敏感信息规则"})
|
|
continue
|
|
item.text = text
|
|
item.title = _title(text, item.title)
|
|
else:
|
|
item.safety["warnings"].append("文档转换不可用,仅保留原始附件")
|
|
allowed.append(item)
|
|
scan.files = allowed
|
|
digest_input = "\n".join(f"{item.path}\0{item.sha256}" for item in scan.files).encode("utf-8")
|
|
scan.source_digest = f"sha256:{hashlib.sha256(digest_input).hexdigest()}"
|
|
return scan
|