deerflow-code/offline-backend-20260512/backend/packages/harness/deerflow/config/workflow_config.py
2026-09-07 18:24:55 +08:00

88 lines
3.8 KiB
Python

"""Workflow Studio configuration (limits, egress policy, sandbox, iframe origins)."""
from __future__ import annotations
from pydantic import BaseModel, Field
class WorkflowEmbedConfig(BaseModel):
"""iframe ticket / origin policy."""
allowed_origins: list[str] = Field(
default_factory=list,
description="Allowed parent origins for workflow embed tickets (exact match)",
)
ticket_ttl_seconds: int = Field(default=60, ge=5, le=600)
class WorkflowHttpPolicyConfig(BaseModel):
"""Egress policy for the ``http`` node.
Default-deny on private address space. ``allowed_hosts`` is an explicit
allowlist (exact host or ``.suffix`` match); when it is empty every public
host is allowed but private/loopback/link-local targets are still refused,
which is what stops SSRF against cloud metadata and internal services.
"""
enabled: bool = Field(default=True)
allowed_hosts: list[str] = Field(default_factory=list)
blocked_hosts: list[str] = Field(default_factory=list)
allow_private_networks: bool = Field(
default=False,
description="Allow loopback/private/link-local targets (only for trusted intranet deployments)",
)
allowed_ports: list[int] = Field(default_factory=lambda: [80, 443, 8080, 8443])
max_response_bytes: int = Field(default=2_097_152, ge=1024)
max_redirects: int = Field(default=0, ge=0, le=5)
timeout_seconds: int = Field(default=60, ge=1)
class WorkflowSqlPolicyConfig(BaseModel):
"""Read-only policy for the ``sql_read`` node."""
enabled: bool = Field(default=True)
max_rows: int = Field(default=1000, ge=1)
statement_timeout_seconds: int = Field(default=30, ge=1)
max_cell_chars: int = Field(default=4000, ge=64)
class WorkflowCodeNodeConfig(BaseModel):
"""``code`` node execution. Off by default: the local runner is a hardened
subprocess, not a container, so operators must opt in knowingly."""
enabled: bool = Field(default=False)
timeout_seconds: int = Field(default=30, ge=1, le=600)
max_source_chars: int = Field(default=20_000, ge=64)
max_output_chars: int = Field(default=100_000, ge=256)
memory_limit_mb: int = Field(default=512, ge=64)
allow_network: bool = Field(default=False)
class WorkflowRetentionConfig(BaseModel):
event_retention_days: int = Field(default=30, ge=1)
run_retention_days: int = Field(default=90, ge=1)
class WorkflowConfig(BaseModel):
"""System-level workflow limits and policies."""
enabled: bool = Field(default=True, description="Master switch for workflow studio APIs")
max_steps: int = Field(default=500, ge=1)
max_loop_iterations: int = Field(default=20, ge=1)
max_parallelism: int = Field(default=16, ge=1)
run_timeout_seconds: int = Field(default=3600, ge=1)
# LangGraph counts a model → tool turn as roughly two graph steps. Deep
# research agents therefore routinely exceed a chat-sized limit of 60.
agent_recursion_limit: int = Field(default=250, ge=20, le=1000)
# Deep Research report nodes are durable jobs and can legitimately need
# longer than a normal chat/tool step. Individual graph node timeouts stay
# opt-in, but the system cap must not silently kill those report jobs.
node_timeout_seconds: int = Field(default=1800, ge=1)
max_concurrent_runs_per_user: int = Field(default=5, ge=1)
lease_ttl_seconds: int = Field(default=60, ge=10)
embed: WorkflowEmbedConfig = Field(default_factory=WorkflowEmbedConfig)
http: WorkflowHttpPolicyConfig = Field(default_factory=WorkflowHttpPolicyConfig)
sql: WorkflowSqlPolicyConfig = Field(default_factory=WorkflowSqlPolicyConfig)
code: WorkflowCodeNodeConfig = Field(default_factory=WorkflowCodeNodeConfig)
retention: WorkflowRetentionConfig = Field(default_factory=WorkflowRetentionConfig)