deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/tools/builtins/skill_tools.py
2026-09-07 18:24:55 +08:00

391 lines
16 KiB
Python

"""skill_view and skill_list — read-only tools for agent skill introspection."""
from __future__ import annotations
import json
import logging
from langchain.tools import ToolRuntime, tool
logger = logging.getLogger(__name__)
_ROUNDTABLE_COORDINATOR_AGENT_ID = "roundtable-coordinator"
COORDINATOR_ONLY_SKILL_NAMES = frozenset({"agent_orchestration", "agent-orchestration"})
def can_access_coordinator_only_skill(agent_id: str | None, skill_name: str) -> bool:
"""Return whether this runtime identity may discover/read a reserved orchestration skill."""
normalized = (skill_name or "").strip().lower()
return normalized not in COORDINATOR_ONLY_SKILL_NAMES or agent_id == _ROUNDTABLE_COORDINATOR_AGENT_ID
def is_coordinator_only_skill_path(path: str) -> bool:
"""Recognize reserved skill directories in virtual/container paths."""
parts = {part.strip().lower() for part in (path or "").replace("\\", "/").split("/") if part}
return bool(parts & COORDINATOR_ONLY_SKILL_NAMES)
def _resolve_active_agent_id(runtime: ToolRuntime | None) -> str | None:
"""Extract the agent_id under which the current tool call is running.
Custom agents pass their id via either the ``context`` dict or the
``configurable`` section of the LangGraph runtime config. The default
lead agent runs without an explicit agent_id — callers should treat
a None return as "no per-agent skills filter applies".
"""
if runtime is None:
return None
try:
if getattr(runtime, "context", None):
ctx_id = runtime.context.get("agent_id") or runtime.context.get("agent_name")
if ctx_id:
return str(ctx_id)
cfg = getattr(runtime, "config", None)
if isinstance(cfg, dict):
configurable = cfg.get("configurable", {})
cid = configurable.get("agent_id") or configurable.get("agent_name")
if cid:
return str(cid)
except Exception:
return None
return None
def _load_agent_skill_whitelist(agent_id: str | None, extra_skills: list[str] | None = None) -> set[str] | None:
"""Return the explicit skill name set declared in an agent's config.yaml.
Semantics:
- ``None`` returned → no agent_id, or the agent's on-disk config is
missing entirely — caller should fall back to "all skills visible"
(default lead agent behavior).
- ``set()`` (empty) → custom agent with ``skills: []`` declared **or**
with no ``skills`` key at all — visibility is intentionally empty.
This mirrors the prompt layer (``agents_config.AgentConfig.skills``:
``None``/``[]`` both mean "no skills" for custom agents), so an agent
whose config omits ``skills`` can no longer enumerate every skill
through this tool.
- Non-empty set → only these names are visible to this agent.
``extra_skills`` only widens the **deep-research collector** allowlist
for one run (report-structure 检索来源). Other agents ignore it.
"""
if not agent_id:
return None
try:
from deerflow.config.agents_config import load_agent_config
cfg = load_agent_config(agent_id)
if cfg is None:
return None
allowed = {
name
for name in (cfg.skills or [])
if can_access_coordinator_only_skill(agent_id, name)
}
except Exception:
logger.debug("Failed to load agent config for skill_list filter", exc_info=True)
return None
if extra_skills:
from deerflow.agents.deep_research.retrieval import merge_collector_skill_allowlist
merged = merge_collector_skill_allowlist(agent_id, list(allowed), extra_skills)
return set(merged or [])
return allowed
def _resolve_extra_skills(runtime: ToolRuntime | None) -> list[str]:
"""Extra skill names from this run's context (collector 检索来源)."""
if runtime is None:
return []
try:
from deerflow.agents.deep_research.retrieval import runtime_extra_skills
if getattr(runtime, "context", None):
extras = runtime_extra_skills(runtime.context)
if extras:
return extras
cfg = getattr(runtime, "config", None)
if isinstance(cfg, dict):
return runtime_extra_skills(cfg.get("configurable", {}) or {})
except Exception:
return []
return []
def _resolve_effective_user_id() -> str | None:
"""当前请求/线程绑定的用户 id;解析失败时返回 None。"""
try:
from deerflow.runtime.user_context import get_effective_user_id
return get_effective_user_id()
except Exception:
logger.debug("Failed to resolve effective user id for skill_list", exc_info=True)
return None
def _load_owned_skill_names(user_id: str | None) -> set[str] | None:
"""返回 ``owner_user_id == user_id`` 的自定义技能名集合。
用于默认 lead agent 的可见性过滤:只算"用户自己拥有的"自定义技能,不含
其他用户发布的。返回 ``None`` 表示 ``SkillStore`` 不可用(未配置数据库
等)—— 调用方应降级为"放行全部自定义技能",避免离线无库部署下技能凭空
消失。
"""
if not user_id:
return None
try:
from deerflow.persistence.engine import get_session_factory, run_db_blocking
from deerflow.persistence.skills import make_skill_store
factory = get_session_factory()
if factory is None:
return None
store = make_skill_store(factory)
records = run_db_blocking(store.list_visible(user_id))
return {r["name"] for r in records if r.get("owner_user_id") == user_id}
except Exception:
logger.debug("Failed to load owned skill names for skill_list filter", exc_info=True)
return None
def skill_view_impl(name: str, agent_id: str | None, extra_skills: list[str] | None = None) -> str:
"""Read a skill's SKILL.md, honoring the agent's config.yaml whitelist.
Custom agents may only read skills on their whitelist — same filter as
``skill_list``, so a restricted agent cannot bypass its allowlist by
guessing skill names. Replies "not found" rather than "forbidden" to
avoid leaking which skills exist.
"""
if not can_access_coordinator_only_skill(agent_id, name):
return f"Skill '{name}' not found."
from deerflow.skills.storage import get_or_new_skill_storage
from deerflow.skills.usage import bump_view
allowed = _load_agent_skill_whitelist(agent_id, extra_skills)
if allowed is not None and name not in allowed:
return f"Skill '{name}' not found."
storage = get_or_new_skill_storage()
try:
if storage.custom_skill_exists(name):
content = storage.read_custom_skill(name)
elif storage.public_skill_exists(name):
skill_file = storage.get_skills_root_path()
from pathlib import Path
from deerflow.skills.types import SKILL_MD_FILE
path = Path(skill_file) / "public" / name / SKILL_MD_FILE
if not path.exists():
return f"Skill '{name}' not found."
content = path.read_text(encoding="utf-8")
else:
return f"Skill '{name}' not found."
except Exception as e:
logger.debug("skill_view failed for %r: %s", name, e)
return f"Error reading skill '{name}': {e}"
try:
bump_view(name)
except Exception:
pass
return content
@tool("skill_view")
def skill_view_tool(name: str, runtime: ToolRuntime) -> str:
"""Read the full SKILL.md content of a skill by name.
Args:
name: The skill name (hyphen-case, e.g. 'my-skill').
"""
return skill_view_impl(name, _resolve_active_agent_id(runtime), _resolve_extra_skills(runtime))
def _resolve_skills_container_base_path() -> str:
"""容器内技能挂载根路径(用于回传 location,供 agent read_file)。"""
try:
from deerflow.config import get_app_config
return get_app_config().skills.container_path
except Exception:
return "/mnt/skills"
def _score_skill(query_terms: list[str], haystack: str) -> int:
"""对单个技能按关键词命中计分(纯关键词、离线、确定性)。"""
score = 0
for term in query_terms:
if term and term in haystack:
score += 1
return score
def search_skills_impl(query: str, agent_id: str | None, top_n: int = 5, extra_skills: list[str] | None = None) -> str:
"""按关键词检索可用技能,返回 top-N 的 name/description/category/location。
与 ``skill_list`` 共享可见性过滤(自定义 agent 走白名单;默认 lead agent 只
暴露未归档且内置或属于当前用户的技能),并只检索**已启用**的技能 —— 它是
"技能压缩"开启后,agent 调出被压缩(非常驻)技能的主要入口。匹配字段为技能
名与描述。
"""
import re
from deerflow.skills.storage import get_or_new_skill_storage
from deerflow.skills.types import SkillCategory
from deerflow.skills.usage import STATE_ARCHIVED, all_entries
query = (query or "").strip()
if not query:
return "search_skills: empty query. Provide a short keyword query describing the task."
terms = [t for t in re.findall(r"\w+", query.lower()) if t]
# 整串也作为一个匹配项,兼顾中文等不被 \w+ 切分出有意义 token 的查询。
whole = query.lower()
if whole and whole not in terms:
terms.append(whole)
allowed = _load_agent_skill_whitelist(agent_id, extra_skills)
owned: set[str] | None = None
if allowed is None:
owned = _load_owned_skill_names(_resolve_effective_user_id())
try:
top_n = max(1, min(int(top_n or 5), 50))
except (TypeError, ValueError):
top_n = 5
try:
storage = get_or_new_skill_storage()
skills = storage.load_skills(enabled_only=True)
usage = all_entries()
base_path = _resolve_skills_container_base_path()
scored: list[tuple[int, dict]] = []
for skill in skills:
if not can_access_coordinator_only_skill(agent_id, skill.name):
continue
entry = usage.get(skill.name, {})
if allowed is not None:
if skill.name not in allowed:
continue
else:
if entry.get("state") == STATE_ARCHIVED:
continue
if skill.category == SkillCategory.CUSTOM and owned is not None and skill.name not in owned:
continue
haystack = f"{skill.name} {skill.description}".lower()
score = _score_skill(terms, haystack)
if score <= 0:
continue
scored.append((
score,
{
"name": skill.name,
"description": skill.description,
"category": skill.category.value if hasattr(skill.category, "value") else str(skill.category),
"location": skill.get_container_file_path(base_path),
},
))
scored.sort(key=lambda x: x[0], reverse=True)
results = [item for _, item in scored[:top_n]]
if not results:
return f"No skills matched '{query}'. Try a broader keyword, or call skill_list to browse everything available."
return json.dumps(results, ensure_ascii=False, indent=2)
except Exception as e:
logger.debug("search_skills failed: %s", e)
return f"Error searching skills: {e}"
@tool("search_skills")
def search_skills_tool(query: str, runtime: ToolRuntime) -> str:
"""Search available skills by keyword and return the best matches.
Use this to discover skills that are not already described in full in your
system prompt. Returns a JSON array of matches, each with `name`,
`description`, `category`, and `location` — then `read_file` the `location`
of the most relevant match and follow its SKILL.md instructions.
Args:
query: A short keyword query describing the task (e.g. 'pdf table extract').
"""
return search_skills_impl(query, _resolve_active_agent_id(runtime), extra_skills=_resolve_extra_skills(runtime))
@tool("skill_list")
def skill_list_tool(runtime: ToolRuntime) -> str:
"""List skills available to this agent, with metadata and usage stats.
Does NOT read SKILL.md content.
For a **custom agent**, the result is filtered to skills explicitly
declared in the agent's ``.deer-flow/agents/{agent_id}/config.yaml``
under the ``skills:`` field — the agent only sees what it was wired to
use.
For the **default lead agent** (no agent_id in runtime), the result is
filtered to non-archived skills that are either built-in (public) or
owned by the current user. Other users' published custom skills and
archived skills are excluded.
Returns a JSON array with fields: name, description, category, enabled,
state, use_count, view_count, patch_count, last_activity_at, pinned,
source.
"""
return skill_list_impl(_resolve_active_agent_id(runtime), _resolve_extra_skills(runtime))
def skill_list_impl(agent_id: str | None, extra_skills: list[str] | None = None) -> str:
"""List skills visible to ``agent_id`` (None = default lead agent)."""
from deerflow.skills.storage import get_or_new_skill_storage
from deerflow.skills.types import SkillCategory
from deerflow.skills.usage import STATE_ARCHIVED, all_entries, latest_activity_at
allowed = _load_agent_skill_whitelist(agent_id, extra_skills)
# 默认 lead agent 没有 per-agent 白名单 —— 改为只暴露"未归档"且"内置或属于
# 当前用户"的技能。owned 为 None 表示 SkillStore 不可用,降级放行全部自定义
# 技能(见 _load_owned_skill_names)。
owned: set[str] | None = None
if allowed is None:
owned = _load_owned_skill_names(_resolve_effective_user_id())
try:
storage = get_or_new_skill_storage()
skills = storage.load_skills(enabled_only=False)
usage = all_entries()
result = []
for skill in skills:
if not can_access_coordinator_only_skill(agent_id, skill.name):
continue
entry = usage.get(skill.name, {})
if allowed is not None:
# 自定义 agent:严格按 config.yaml 的 skills 白名单过滤。
if skill.name not in allowed:
continue
else:
# 默认 agent:跳过已归档技能。
if entry.get("state") == STATE_ARCHIVED:
continue
# 自定义技能仅当属于当前用户时可见;内置 public 技能始终可见。
if skill.category == SkillCategory.CUSTOM and owned is not None and skill.name not in owned:
continue
result.append({
"name": skill.name,
"description": skill.description,
"category": skill.category.value if hasattr(skill.category, "value") else str(skill.category),
"enabled": skill.enabled,
"state": entry.get("state", "active"),
"use_count": entry.get("use_count", 0),
"view_count": entry.get("view_count", 0),
"patch_count": entry.get("patch_count", 0),
"last_activity_at": latest_activity_at(skill.name),
"pinned": entry.get("pinned", False),
"source": entry.get("source", "upload"),
})
return json.dumps(result, ensure_ascii=False, indent=2)
except Exception as e:
logger.debug("skill_list failed: %s", e)
return f"Error listing skills: {e}"