"""Safe, deterministic scanner for the knowledge-bearing part of a skill.""" from __future__ import annotations import csv import hashlib import json import re import zipfile from dataclasses import dataclass, field from pathlib import Path from typing import Any CODE_SUFFIXES = { ".py", ".pyc", ".pyo", ".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".sh", ".bash", ".zsh", ".fish", ".bat", ".cmd", ".ps1", ".psm1", ".java", ".kt", ".kts", ".go", ".rs", ".c", ".cc", ".cpp", ".h", ".hpp", ".cs", ".php", ".rb", ".pl", ".lua", ".swift", ".scala", ".sql", ".vue", ".svelte", } TEXT_SUFFIXES = {".md", ".mdx", ".txt", ".json", ".yaml", ".yml", ".toml", ".ini", ".csv", ".tsv"} BINARY_KNOWLEDGE_SUFFIXES = {".xlsx", ".xls", ".pdf", ".docx", ".pptx", ".png", ".jpg", ".jpeg", ".webp", ".gif", ".bmp", ".zip"} EXCLUDED_DIRS = {".git", "node_modules", "__pycache__", ".pytest_cache", "dist", "build", ".mypy_cache", ".ruff_cache", ".address-edits"} EXCLUDED_FILENAMES = { "package-lock.json", "pnpm-lock.yaml", "yarn.lock", "bun.lockb", "poetry.lock", "uv.lock", "pipfile.lock", "cargo.lock", "composer.lock", } SENSITIVE_NAME_RE = re.compile( r"(^|[._-])(\.env|id_rsa|id_dsa|id_ecdsa|id_ed25519|private[_-]?key|credentials?|secrets?)([._-]|$)", re.IGNORECASE, ) SECRET_CONTENT_RE = re.compile(r"(?im)^\s*(?:api[_-]?key|access[_-]?key|secret(?:[_-]?key)?|password|token|private[_-]?key)\s*[:=]\s*[^\s]{8,}\s*$") PEM_PRIVATE_RE = re.compile(r"-----BEGIN (?:RSA |EC |OPENSSH )?PRIVATE KEY-----") FENCE_RE = re.compile(r"(?ms)^\s*(```|~~~).*?^\s*\1\s*$") MARKDOWN_LINK_RE = re.compile(r"\[[^\]]+\]\(([^)#?]+)(?:#[^)]*)?\)") MAX_TEXT_BYTES = 2 * 1024 * 1024 MAX_ARCHIVE_FILES = 1_000 MAX_ARCHIVE_ENTRY_BYTES = 20 * 1024 * 1024 MAX_ARCHIVE_TOTAL_BYTES = 200 * 1024 * 1024 @dataclass(slots=True) class ScannedFile: path: str kind: str sha256: str size: int text: str = "" title: str = "" references: list[dict[str, Any]] = field(default_factory=list) safety: dict[str, Any] = field(default_factory=lambda: {"status": "allowed", "warnings": []}) def manifest(self) -> dict[str, Any]: return { "path": self.path, "kind": self.kind, "sha256": self.sha256, "size": self.size, "text_digest": f"sha256:{hashlib.sha256(self.text.encode('utf-8')).hexdigest()}" if self.text else None, "title": self.title, "references": self.references, "safety": self.safety, } @dataclass(slots=True) class ScanResult: skill_name: str source_root: str source_digest: str files: list[ScannedFile] blocked_files: list[dict[str, str]] skipped_code_count: int skipped_other_count: int @property def manifest(self) -> list[dict[str, Any]]: return [item.manifest() for item in self.files] + [{"path": item["path"], "kind": "blocked", "safety": {"status": "blocked", "warnings": [item["reason"]]}} for item in self.blocked_files] @property def counts(self) -> dict[str, int]: return { "knowledge_files": len(self.files), "blocked_files": len(self.blocked_files), "skipped_code_files": self.skipped_code_count, "skipped_other_files": self.skipped_other_count, } def _safe_relative(path: Path, root: Path) -> str: resolved = path.resolve() resolved_root = root.resolve() try: relative = resolved.relative_to(resolved_root) except ValueError as exc: raise ValueError("Skill file escapes its source root") from exc return relative.as_posix() def _strip_markdown_code(text: str) -> str: return FENCE_RE.sub("\n[示例代码已按安全策略省略]\n", text) def _read_text(path: Path, suffix: str) -> str: raw = path.read_bytes() if len(raw) > MAX_TEXT_BYTES: raise ValueError("text file exceeds 2 MiB parsing limit") text = raw.decode("utf-8-sig", errors="replace") if suffix in {".md", ".mdx"}: return _strip_markdown_code(text) if suffix in {".csv", ".tsv"}: delimiter = "\t" if suffix == ".tsv" else "," rows = list(csv.reader(text.splitlines(), delimiter=delimiter)) return "\n".join(" | ".join(cell.strip() for cell in row) for row in rows[:500]) if suffix == ".json": try: return json.dumps(json.loads(text), ensure_ascii=False, indent=2) except json.JSONDecodeError: return text return text def _title(text: str, fallback: str) -> str: match = re.search(r"(?m)^#\s+(.+?)\s*$", text) return match.group(1).strip() if match else fallback def _kind(suffix: str) -> str: return { ".md": "markdown", ".mdx": "markdown", ".txt": "text", ".csv": "table", ".tsv": "table", ".xlsx": "spreadsheet", ".xls": "spreadsheet", ".pdf": "pdf", ".docx": "office", ".pptx": "office", ".png": "image", ".jpg": "image", ".jpeg": "image", ".webp": "image", ".gif": "image", ".bmp": "image", ".zip": "archive", }.get(suffix, "structured_data") def _zip_members(path: Path, archive_relative: str) -> tuple[list[ScannedFile], list[dict[str, str]], int, int]: files: list[ScannedFile] = [] blocked: list[dict[str, str]] = [] skipped_code = 0 skipped_other = 0 with zipfile.ZipFile(path) as archive: members = archive.infolist() if len(members) > MAX_ARCHIVE_FILES: return [], [{"path": archive_relative, "reason": "压缩包文件数超过安全上限"}], 0, 0 total_size = sum(member.file_size for member in members) if total_size > MAX_ARCHIVE_TOTAL_BYTES: return [], [{"path": archive_relative, "reason": "压缩包展开体积超过安全上限"}], 0, 0 for member in members: member_path = Path(member.filename.replace("\\", "/")) virtual_path = f"{archive_relative}!/{member_path.as_posix()}" unix_mode = member.external_attr >> 16 if member.is_dir(): continue if member_path.is_absolute() or ".." in member_path.parts or (unix_mode & 0o170000) == 0o120000: blocked.append({"path": virtual_path, "reason": "压缩包包含路径穿越或符号链接"}) continue suffix = member_path.suffix.lower() lower_name = member_path.name.lower() if lower_name in EXCLUDED_FILENAMES or suffix in CODE_SUFFIXES: skipped_code += 1 continue if SENSITIVE_NAME_RE.search(lower_name) or suffix in {".pem", ".key", ".p12", ".pfx"}: blocked.append({"path": virtual_path, "reason": "疑似密钥或敏感配置文件"}) continue if suffix not in TEXT_SUFFIXES or member.file_size > MAX_ARCHIVE_ENTRY_BYTES: skipped_other += 1 continue raw = archive.read(member) text = raw.decode("utf-8-sig", errors="replace") if suffix in {".md", ".mdx"}: text = _strip_markdown_code(text) if SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text): blocked.append({"path": virtual_path, "reason": "内容命中高置信敏感信息规则"}) continue files.append( ScannedFile( path=virtual_path, kind=_kind(suffix), sha256=hashlib.sha256(raw).hexdigest(), size=len(raw), text=text, title=_title(text, member_path.stem), ) ) return files, blocked, skipped_code, skipped_other def scan_skill(skill_name: str, skill_root: Path) -> ScanResult: """Scan one already-resolved skill directory without following symlinks.""" root = skill_root.resolve() if not root.is_dir(): raise FileNotFoundError(f"Skill '{skill_name}' source directory does not exist") files: list[ScannedFile] = [] blocked: list[dict[str, str]] = [] skipped_code = 0 skipped_other = 0 for path in sorted(root.rglob("*"), key=lambda item: item.as_posix().lower()): if path.is_symlink() or not path.is_file(): continue relative = _safe_relative(path, root) if any(part.lower() in EXCLUDED_DIRS for part in Path(relative).parts[:-1]): continue lower_name = path.name.lower() suffix = path.suffix.lower() if lower_name in EXCLUDED_FILENAMES or suffix in CODE_SUFFIXES: skipped_code += 1 continue if SENSITIVE_NAME_RE.search(lower_name) or suffix in {".pem", ".key", ".p12", ".pfx"}: blocked.append({"path": relative, "reason": "疑似密钥或敏感配置文件"}) continue if suffix not in TEXT_SUFFIXES | BINARY_KNOWLEDGE_SUFFIXES: skipped_other += 1 continue # Never upload an archive wholesale: it may contain excluded code or a # blocked secret. Only independently validated knowledge members enter # the manifest and digest. if suffix == ".zip": try: nested, nested_blocked, nested_code, nested_other = _zip_members(path, relative) files.extend(nested) blocked.extend(nested_blocked) skipped_code += nested_code skipped_other += nested_other except (OSError, zipfile.BadZipFile): blocked.append({"path": relative, "reason": "压缩包损坏或格式不受支持"}) continue raw = path.read_bytes() digest = hashlib.sha256(raw).hexdigest() text = "" warnings: list[str] = [] if suffix in TEXT_SUFFIXES: try: text = _read_text(path, suffix) except ValueError as exc: warnings.append(str(exc)) if text and (SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text)): blocked.append({"path": relative, "reason": "内容命中高置信敏感信息规则"}) continue references = [] if suffix in {".md", ".mdx"}: for target in MARKDOWN_LINK_RE.findall(text): normalized = target.replace("\\", "/").lstrip("./") if normalized: references.append({"target_path": normalized, "evidence": "markdown_link", "confidence": 1.0}) files.append( ScannedFile( path=relative, kind=_kind(suffix), sha256=digest, size=len(raw), text=text, title=_title(text, path.stem), references=references, safety={"status": "allowed", "warnings": warnings}, ) ) digest_input = "\n".join(f"{item.path}\0{item.sha256}" for item in files).encode("utf-8") source_digest = f"sha256:{hashlib.sha256(digest_input).hexdigest()}" return ScanResult( skill_name=skill_name, source_root=str(root), source_digest=source_digest, files=files, blocked_files=blocked, skipped_code_count=skipped_code, skipped_other_count=skipped_other, ) async def enrich_converted_documents(scan: ScanResult) -> ScanResult: """Extract Office/PDF/spreadsheet text without modifying the skill tree.""" from deerflow.utils.file_conversion import CONVERTIBLE_EXTENSIONS, extract_file_to_markdown allowed: list[ScannedFile] = [] root = Path(scan.source_root) for item in scan.files: if Path(item.path).suffix.lower() in CONVERTIBLE_EXTENSIONS: text = await extract_file_to_markdown(root / item.path) if text: text = _strip_markdown_code(text) if SECRET_CONTENT_RE.search(text) or PEM_PRIVATE_RE.search(text): scan.blocked_files.append({"path": item.path, "reason": "转换文本命中高置信敏感信息规则"}) continue item.text = text item.title = _title(text, item.title) else: item.safety["warnings"].append("文档转换不可用,仅保留原始附件") allowed.append(item) scan.files = allowed digest_input = "\n".join(f"{item.path}\0{item.sha256}" for item in scan.files).encode("utf-8") scan.source_digest = f"sha256:{hashlib.sha256(digest_input).hexdigest()}" return scan