deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/tools/tools.py
2026-09-07 18:24:55 +08:00

315 lines
15 KiB
Python

import logging
from langchain.tools import BaseTool
from deerflow.config import get_app_config
from deerflow.config.app_config import AppConfig
from deerflow.reflection import resolve_variable
from deerflow.sandbox.security import is_host_bash_allowed
from deerflow.tools.builtins import agent_orchestration_tool, ask_clarification_tool, memory_tool, present_file_tool, scheduled_task_tool, task_tool, view_image_tool
from deerflow.tools.builtins.tool_search import reset_deferred_registry
logger = logging.getLogger(__name__)
BROWSER_CONTEXT_TOOL_USE = "deerflow.tools.builtins.browser_context_tool:browser_fetch_page_tool"
BROWSER_ACT_TOOL_USE = "deerflow.tools.builtins.browser_act_tool:browser_act_tool"
BROWSER_CONTEXT_TOOL_USES = {BROWSER_CONTEXT_TOOL_USE, BROWSER_ACT_TOOL_USE}
BROWSER_CONTEXT_TOOL_NAMES = {"browser_fetch_page", "browser_act"}
BUILTIN_TOOLS = [
present_file_tool,
ask_clarification_tool,
]
SUBAGENT_TOOLS = [
task_tool,
# task_status_tool is no longer exposed to LLM (backend handles polling internally)
]
def _is_host_bash_tool(tool: object) -> bool:
"""Return True if the tool config represents a host-bash execution surface."""
group = getattr(tool, "group", None)
use = getattr(tool, "use", None)
if group == "bash":
return True
if use == "deerflow.sandbox.tools:bash_tool":
return True
return False
def _is_browser_context_tool(tool: object) -> bool:
"""Return True if the tool config represents the browser extension bridge."""
use = getattr(tool, "use", None)
name = getattr(tool, "name", None)
return use in BROWSER_CONTEXT_TOOL_USES or name in BROWSER_CONTEXT_TOOL_NAMES
def _enforce_agent_orchestration_scope(
tools: list[BaseTool],
*,
allow_agent_orchestration: bool,
) -> list[BaseTool]:
"""Fail closed: only an explicitly authorized coordinator may receive orchestration."""
if allow_agent_orchestration:
return tools
return [tool for tool in tools if tool.name != "agent_orchestration"]
def get_available_tools(
groups: list[str] | None = None,
include_mcp: bool = True,
model_name: str | None = None,
subagent_enabled: bool = False,
is_scheduled_run: bool = False,
excluded_tools: list[str] | None = None,
*,
allow_agent_orchestration: bool = False,
app_config: AppConfig | None = None,
) -> list[BaseTool]:
"""Get all available tools from config.
Note: MCP tools should be initialized at application startup using
`initialize_mcp_tools()` from deerflow.mcp module.
Args:
groups: Optional list of tool groups to filter by.
include_mcp: Whether to include tools from MCP servers (default: True).
model_name: Optional model name to determine if vision tools should be included.
subagent_enabled: Whether to include subagent tools (task, task_status).
excluded_tools: Optional list of tool names to exclude from the result
(applied AFTER all other resolution). Takes precedence over
``app_config.excluded_tools``.
allow_agent_orchestration: Explicit coordinator-only capability gate. Defaults to
``False`` so ordinary agents and all indirect callers fail closed.
Returns:
List of available tools.
"""
config = app_config or get_app_config()
selected_tool_configs = [tool for tool in config.tools if groups is None or tool.group in groups]
disabled_tool_names = [tool.name for tool in selected_tool_configs if not tool.enabled]
if disabled_tool_names:
logger.info("Skipping disabled configured tool(s): %s", disabled_tool_names)
tool_configs = [tool for tool in selected_tool_configs if tool.enabled]
browser_context_config = getattr(config, "browser_context", None)
if not getattr(browser_context_config, "enabled", False):
tool_configs = [tool for tool in tool_configs if not _is_browser_context_tool(tool)]
elif not getattr(browser_context_config, "actions_enabled", True):
# browser_context is on, but interactive page actions are disabled — keep
# the read-only browser_fetch_page tool, drop only browser_act.
tool_configs = [
tool
for tool in tool_configs
if getattr(tool, "use", None) != BROWSER_ACT_TOOL_USE and getattr(tool, "name", None) != "browser_act"
]
# Do not expose host bash by default when LocalSandboxProvider is active.
if not is_host_bash_allowed(config):
tool_configs = [tool for tool in tool_configs if not _is_host_bash_tool(tool)]
loaded_tools_raw = [(cfg, resolve_variable(cfg.use, BaseTool)) for cfg in tool_configs]
# Warn when the config ``name`` field and the tool object's ``.name``
# attribute diverge — this mismatch is the root cause of issue #1803 where
# the LLM receives one name in its tool schema but the runtime router
# recognises a different name, producing "not a valid tool" errors.
for cfg, loaded in loaded_tools_raw:
if cfg.name != loaded.name:
logger.warning(
"Tool name mismatch: config name %r does not match tool .name %r (use: %s). The tool's own .name will be used for binding.",
cfg.name,
loaded.name,
cfg.use,
)
loaded_tools = [t for _, t in loaded_tools_raw]
# Conditionally add tools based on config
builtin_tools = BUILTIN_TOOLS.copy()
if allow_agent_orchestration:
builtin_tools.append(agent_orchestration_tool)
if not is_scheduled_run:
builtin_tools.append(scheduled_task_tool)
# Memory V2:记忆与 builtin provider 都启用时,暴露 memory 工具给 LLM 主动写入
memory_config = getattr(config, "memory", None)
if memory_config is not None and getattr(memory_config, "enabled", False):
builtin_config = getattr(memory_config, "builtin", None)
if builtin_config is not None and getattr(builtin_config, "enabled", False):
builtin_tools.append(memory_tool)
logger.info("Including memory tool (Memory V2 builtin provider enabled)")
# Hindsight 在 tools / hybrid 模式下,额外暴露 3 个 Hindsight 工具
if getattr(memory_config, "provider", "builtin") == "hindsight":
hindsight_config = getattr(memory_config, "hindsight", None)
memory_mode = getattr(hindsight_config, "memory_mode", "context") if hindsight_config else "context"
if memory_mode in ("tools", "hybrid"):
from deerflow.tools.builtins.hindsight_tools import HINDSIGHT_TOOLS
builtin_tools.extend(HINDSIGHT_TOOLS)
logger.info("Including Hindsight tools (memory_mode=%s)", memory_mode)
skill_evolution_config = getattr(config, "skill_evolution", None)
# skill_list / skill_view 是只读工具(发现、查看技能),与"自我进化"写能力无关,
# 始终提供给 agent。
from deerflow.tools.builtins.skill_tools import skill_list_tool, skill_view_tool
builtin_tools.extend([skill_list_tool, skill_view_tool])
# search_skills —— 仅当管理员开启"技能压缩"时提供。压缩开启后,非常驻技能不再
# 全文注入系统提示词,agent 通过该关键词检索工具按需调出(纯关键词、离线、确定性)。
try:
from deerflow.config.system_settings import get_skill_compression_settings
if get_skill_compression_settings().enabled:
from deerflow.tools.builtins.skill_tools import search_skills_tool
builtin_tools.append(search_skills_tool)
except Exception: # noqa: BLE001 - 读取设置失败不能阻断 agent 构建
logger.debug("Failed to evaluate skill compression setting for search_skills tool", exc_info=True)
# read_peer_delivery —— 圆桌席位「按需读取前序席位完整交付」的只读工具(从 run context
# 取网关注入的全文,非圆桌场景为空 → 工具自述「无」)。始终提供(只读、无副作用);圆桌
# 总控/报告角色由 run policy 显式排除,避免干扰它们。
from deerflow.tools.builtins.roundtable_peers_tool import read_peer_delivery_tool
builtin_tools.append(read_peer_delivery_tool)
# Conditional LLMWiki mode: the tool is absent when no effective WeKnora
# address is configured, keeping legacy deployments behavior-identical.
try:
from deerflow.integrations.weknora.runtime import get_resolved_llmwiki_runtime
if get_resolved_llmwiki_runtime(config).weknora_enabled:
from deerflow.tools.builtins.llmwiki_search_tool import llmwiki_search_tool
builtin_tools.append(llmwiki_search_tool)
except Exception: # noqa: BLE001 - configuration/store failures must not block agent creation
logger.debug("Failed to evaluate the conditional LLMWiki search tool", exc_info=True)
# skill_manage 能创建/修改/删除技能,属于"自我进化"写能力,按 skill_evolution.enabled
# 门控。该开关用户可在 per-user 配置里覆盖,所以按"生效配置(含 per-user 覆盖)"判断 ——
# Gateway 每次 run 重建 agent,用户改了配置下次 run 即生效;解析失败退回全局 config.yaml。
try:
from deerflow.config.skill_evolution_runtime import load_skill_evolution_config_for_user
from deerflow.runtime.user_context import get_effective_user_id
skill_evolution_config = load_skill_evolution_config_for_user(get_effective_user_id())
except Exception: # noqa: BLE001 - 解析失败不能阻断 agent 构建
skill_evolution_config = getattr(config, "skill_evolution", None)
if getattr(skill_evolution_config, "enabled", False):
from deerflow.tools.skill_manage_tool import skill_manage_tool
builtin_tools.append(skill_manage_tool)
# Add subagent tools only if enabled via runtime parameter
if subagent_enabled:
builtin_tools.extend(SUBAGENT_TOOLS)
logger.info("Including subagent tools (task)")
# If no model_name specified, use the first model (default)
if model_name is None and config.models:
model_name = config.models[0].name
# Add view_image_tool when vision is available — either the main model
# supports vision natively, or a dedicated vision model is configured
# (vision.model_name) so recognition can be delegated to it.
from deerflow.models.vision import vision_enabled
if vision_enabled(model_name, app_config=config):
builtin_tools.append(view_image_tool)
logger.info(f"Including view_image_tool for model '{model_name}' (vision available)")
# Get cached MCP tools if enabled
# NOTE: We use ExtensionsConfig.from_file() instead of config.extensions
# to always read the latest configuration from disk. This ensures that changes
# made through the Gateway API (which runs in a separate process) are immediately
# reflected when loading MCP tools.
mcp_tools = []
# Reset deferred registry upfront to prevent stale state from previous calls
reset_deferred_registry()
if include_mcp:
try:
from deerflow.config.extensions_config import ExtensionsConfig
from deerflow.mcp.cache import get_cached_mcp_tools
extensions_config = ExtensionsConfig.from_file()
if extensions_config.get_enabled_mcp_servers():
mcp_tools = get_cached_mcp_tools()
if mcp_tools:
logger.info(f"Using {len(mcp_tools)} cached MCP tool(s)")
# When tool_search is enabled, register MCP tools in the
# deferred registry and add tool_search to builtin tools.
if config.tool_search.enabled:
from deerflow.tools.builtins.tool_search import DeferredToolRegistry, set_deferred_registry
from deerflow.tools.builtins.tool_search import tool_search as tool_search_tool
registry = DeferredToolRegistry()
for t in mcp_tools:
registry.register(t)
set_deferred_registry(registry)
builtin_tools.append(tool_search_tool)
logger.info(f"Tool search active: {len(mcp_tools)} tools deferred")
except ImportError:
logger.warning("MCP module not available. Install 'langchain-mcp-adapters' package to enable MCP tools.")
except Exception as e:
logger.error(f"Failed to get cached MCP tools: {e}")
# Add invoke_acp_agent tool if any ACP agents are configured
acp_tools: list[BaseTool] = []
try:
from deerflow.tools.builtins.invoke_acp_agent_tool import build_invoke_acp_agent_tool
if app_config is None:
from deerflow.config.acp_config import get_acp_agents
acp_agents = get_acp_agents()
else:
acp_agents = getattr(config, "acp_agents", {}) or {}
if acp_agents:
acp_tools.append(build_invoke_acp_agent_tool(acp_agents))
logger.info(f"Including invoke_acp_agent tool ({len(acp_agents)} agent(s): {list(acp_agents.keys())})")
except Exception as e:
logger.warning(f"Failed to load ACP tool: {e}")
logger.info(f"Total tools loaded: {len(loaded_tools)}, built-in tools: {len(builtin_tools)}, MCP tools: {len(mcp_tools)}, ACP tools: {len(acp_tools)}")
# Deduplicate by tool name — config-loaded tools take priority, followed by
# built-ins, MCP tools, and ACP tools. Duplicate names cause the LLM to
# receive ambiguous or concatenated function schemas (issue #1803).
all_tools = loaded_tools + builtin_tools + mcp_tools + acp_tools
seen_names: set[str] = set()
unique_tools: list[BaseTool] = []
for t in all_tools:
if t.name not in seen_names:
unique_tools.append(t)
seen_names.add(t.name)
else:
logger.warning(
"Duplicate tool name %r detected and skipped — check your config.yaml and MCP server registrations (issue #1803).",
t.name,
)
# Drop tools the caller (or config) wants excluded. Request-level
# ``excluded_tools`` wins; the config-level list is the static fallback.
effective_excluded = excluded_tools if excluded_tools else getattr(config, "excluded_tools", None) or []
if effective_excluded:
excluded_set = set(effective_excluded)
filtered_tools = [t for t in unique_tools if t.name not in excluded_set]
if len(filtered_tools) < len(unique_tools):
logger.info(
"Excluded %d tool(s) by request/config: %s",
len(unique_tools) - len(filtered_tools),
effective_excluded,
)
unique_tools = filtered_tools
# Defense in depth: a config/MCP tool with the reserved name must not bypass the builtin gate.
return _enforce_agent_orchestration_scope(
unique_tools,
allow_agent_orchestration=allow_agent_orchestration,
)