88 lines
3.8 KiB
Python
88 lines
3.8 KiB
Python
"""Workflow Studio configuration (limits, egress policy, sandbox, iframe origins)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pydantic import BaseModel, Field
|
|
|
|
|
|
class WorkflowEmbedConfig(BaseModel):
|
|
"""iframe ticket / origin policy."""
|
|
|
|
allowed_origins: list[str] = Field(
|
|
default_factory=list,
|
|
description="Allowed parent origins for workflow embed tickets (exact match)",
|
|
)
|
|
ticket_ttl_seconds: int = Field(default=60, ge=5, le=600)
|
|
|
|
|
|
class WorkflowHttpPolicyConfig(BaseModel):
|
|
"""Egress policy for the ``http`` node.
|
|
|
|
Default-deny on private address space. ``allowed_hosts`` is an explicit
|
|
allowlist (exact host or ``.suffix`` match); when it is empty every public
|
|
host is allowed but private/loopback/link-local targets are still refused,
|
|
which is what stops SSRF against cloud metadata and internal services.
|
|
"""
|
|
|
|
enabled: bool = Field(default=True)
|
|
allowed_hosts: list[str] = Field(default_factory=list)
|
|
blocked_hosts: list[str] = Field(default_factory=list)
|
|
allow_private_networks: bool = Field(
|
|
default=False,
|
|
description="Allow loopback/private/link-local targets (only for trusted intranet deployments)",
|
|
)
|
|
allowed_ports: list[int] = Field(default_factory=lambda: [80, 443, 8080, 8443])
|
|
max_response_bytes: int = Field(default=2_097_152, ge=1024)
|
|
max_redirects: int = Field(default=0, ge=0, le=5)
|
|
timeout_seconds: int = Field(default=60, ge=1)
|
|
|
|
|
|
class WorkflowSqlPolicyConfig(BaseModel):
|
|
"""Read-only policy for the ``sql_read`` node."""
|
|
|
|
enabled: bool = Field(default=True)
|
|
max_rows: int = Field(default=1000, ge=1)
|
|
statement_timeout_seconds: int = Field(default=30, ge=1)
|
|
max_cell_chars: int = Field(default=4000, ge=64)
|
|
|
|
|
|
class WorkflowCodeNodeConfig(BaseModel):
|
|
"""``code`` node execution. Off by default: the local runner is a hardened
|
|
subprocess, not a container, so operators must opt in knowingly."""
|
|
|
|
enabled: bool = Field(default=False)
|
|
timeout_seconds: int = Field(default=30, ge=1, le=600)
|
|
max_source_chars: int = Field(default=20_000, ge=64)
|
|
max_output_chars: int = Field(default=100_000, ge=256)
|
|
memory_limit_mb: int = Field(default=512, ge=64)
|
|
allow_network: bool = Field(default=False)
|
|
|
|
|
|
class WorkflowRetentionConfig(BaseModel):
|
|
event_retention_days: int = Field(default=30, ge=1)
|
|
run_retention_days: int = Field(default=90, ge=1)
|
|
|
|
|
|
class WorkflowConfig(BaseModel):
|
|
"""System-level workflow limits and policies."""
|
|
|
|
enabled: bool = Field(default=True, description="Master switch for workflow studio APIs")
|
|
max_steps: int = Field(default=500, ge=1)
|
|
max_loop_iterations: int = Field(default=20, ge=1)
|
|
max_parallelism: int = Field(default=16, ge=1)
|
|
run_timeout_seconds: int = Field(default=3600, ge=1)
|
|
# LangGraph counts a model → tool turn as roughly two graph steps. Deep
|
|
# research agents therefore routinely exceed a chat-sized limit of 60.
|
|
agent_recursion_limit: int = Field(default=250, ge=20, le=1000)
|
|
# Deep Research report nodes are durable jobs and can legitimately need
|
|
# longer than a normal chat/tool step. Individual graph node timeouts stay
|
|
# opt-in, but the system cap must not silently kill those report jobs.
|
|
node_timeout_seconds: int = Field(default=1800, ge=1)
|
|
max_concurrent_runs_per_user: int = Field(default=5, ge=1)
|
|
lease_ttl_seconds: int = Field(default=60, ge=10)
|
|
embed: WorkflowEmbedConfig = Field(default_factory=WorkflowEmbedConfig)
|
|
http: WorkflowHttpPolicyConfig = Field(default_factory=WorkflowHttpPolicyConfig)
|
|
sql: WorkflowSqlPolicyConfig = Field(default_factory=WorkflowSqlPolicyConfig)
|
|
code: WorkflowCodeNodeConfig = Field(default_factory=WorkflowCodeNodeConfig)
|
|
retention: WorkflowRetentionConfig = Field(default_factory=WorkflowRetentionConfig)
|