deerflow-code/offline-backend-20260512/backend/app/gateway/routers/multi_agent.py
2026-09-07 18:24:55 +08:00

2224 lines
111 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""圆桌规划 · Step 2「多智能体研讨」Gateway 路由。
前端页面:``RoundtablePlanningPage`` Step 2
前端模块:``frontend-web/src/roundtable-planning/api/multi-agent.ts``
本模块把原 ``newpython/OUT.py`` 演示编排进 FastAPI Gateway,通过 **HTTP loopback**
(``127.0.0.1`` + 调用方 ``Authorization``)复用既有 ``/api/threads``、
``/api/agents``、``/api/threads/{id}/runs/stream``,不再硬编码 token。
对外接口
--------
1. ``POST /api/multi-agent/init``(SSE)
- 为每个研讨席位(``agent_names``)并行创建 thread,按完成顺序推送
``seat_ready``;
- 动态创建**当次会话**的总控协调智能体(``main_agent_name``,如
``roundtable-coordinator-xxx``),写入席位列表 SOUL + ``agent_orchestration`` skill;
- 结束帧 ``init_done`` 携带 ``thread_ids: { agent_name: thread_id, ... }``,
含协调智能体自身 thread。
2. ``POST /api/multi-agent/run/stream``(SSE)
- ``agent_type=leader``:总控一轮——可能派活、可能澄清、可能共识(空 dispatch);
- ``agent_type=special``:单席位子智能体执行一轮交付。
跨席位「广播」语义
------------------
通过 ``_append_thread_message`` 向其它 thread 的 checkpoint ``messages`` 追加
**只读上下文**的 human 消息(非触发 run),使各席位 LangGraph 状态里能看到
子 agent 完成摘要。总控输入、派活 prompt 和派活说明只留在总控 thread,
绝不广播给席位;前端编排循环仍靠显式多次 ``/run/stream`` 驱动各 agent 真正推理。
Leader 终态帧(前端 ``isFinalStatusFrame``)
-------------------------------------------
- ``status: [[agent_name, task], ...]`` — 派活列表,前端并行 special run;
- ``status: []`` — 无 tool_calls,视为共识,可进 Step 3;
- ``status: "clarification"`` — 总控 ``ask_clarification``,前端暂停编排展示澄清条;
- ``status: "子智能体 X 完成..."`` — special 完成(字符串状态);
- ``status: "invalid_delivery"`` — 席位本轮无可见正文(仅思考 / 断流 / 空串),不算交付;
- ``status: "error"`` — 失败。
共享导出
--------
``intent.py`` / ``recommend.py`` 从此模块 import 底层流式与消息解析函数,
勿在 intent/recommend 中重复实现 loopback 逻辑。
相关文档:``frontend-web/docs/multi-agent-backend-dev.md``
"""
from __future__ import annotations
import asyncio
import json
import logging
import os
import re
import time
import traceback
import uuid
from collections.abc import AsyncIterator
from typing import Any
import httpx
from fastapi import APIRouter, Body, HTTPException, Request
from fastapi.responses import StreamingResponse
from langgraph.checkpoint.base import empty_checkpoint
from app.gateway.roundtable_diag import record_foreground_diag
from app.gateway.roundtable_model_fallback import fallback_model_chain, is_llm_error_message, reason_text
from app.gateway.roundtable_run_policy import merge_skill_stop_names, roundtable_run_policy
from app.gateway.roundtable_seat_skills import append_seat_role_boundary, append_seat_skill_directive
from app.gateway.routers._roundtable_seed import (
ACTION_PLAN_AGENT_ID,
COORDINATOR_AGENT_ID,
DASHBOARD_AGENT_ID,
REPORT_AGENT_ID,
STRUCTURE_AGENT_ID,
SUMMARY_AGENT_ID,
ensure_roundtable_functional_agents,
)
from app.gateway.services import RUN_SCOPE_ORCHESTRATION_CHILD, internal_run_scope_headers
from deerflow.agents.roundtable_orchestrator.delivery import classify_seat_delivery, visible_seat_text
from deerflow.agents.roundtable_orchestrator.extraction_spec import build_summary_directive
from deerflow.config import get_app_config
from deerflow.runtime.user_context import get_effective_user_id
from deerflow.tools.builtins.clarification_utils import resolve_allow_multiple
from deerflow.utils.time import now_iso
logger = logging.getLogger(__name__)
router = APIRouter(prefix="/api/multi-agent", tags=["multi-agent"])
def _stage_for_agent(agent_type: str, agent_name: str) -> str:
"""把一次 /run/stream 的 (agent_type, agent_name) 映射成诊断日志的 stage。
Step3 的四个内置单例(report/summary/dashboard/structure)也走 special 路径,
据 agent_name 归到对应 step3_* 阶段;其余 leader→step2_leader、special→step2_seat。
"""
if agent_name == REPORT_AGENT_ID:
return "step3_report"
if agent_name == SUMMARY_AGENT_ID:
return "step3_summary"
if agent_name == ACTION_PLAN_AGENT_ID:
return "position_action_plan"
if agent_name == DASHBOARD_AGENT_ID:
return "step3_dashboard"
if agent_name == STRUCTURE_AGENT_ID:
return "step3_structure"
return "step2_leader" if agent_type == "leader" else "step2_seat"
# 圆桌(intent / recommend / 席位 / 协调)创建的 thread 都是功能型内部线程,
# 不是用户的普通对话。打上这个 metadata,前端「最近的对话」列表(recent-chat-list.tsx)
# 据 `metadata.thread_type` 过滤掉,与 scheduler / bootstrap 系统会话同一套机制。
_ROUNDTABLE_THREAD_METADATA: dict[str, Any] = {"thread_type": "roundtable", "system": True}
# ---------------------------------------------------------------------------
# 共享:异常诊断器(intent/recommend/multi-agent 共用)
# ---------------------------------------------------------------------------
def _diagnose_exception(exc: BaseException, *, context: str = "") -> dict[str, Any]:
"""把后端异常转成结构化诊断 dict,直接给前端展示用。
设计目标:让运维 / 前端开发不用翻 backend 日志就能看到根因和处置建议。
返回字段:
- code: 短稳定分类(loopback_connect_failed / upstream_connect_failed / ...)
- message: 一行人类可读
- exception_type: 表层异常类名
- exception_repr: repr(exc)
- root_type / root_repr: 沿 __cause__ 链找到的最深异常(若与表层不同)
- hint: 针对常见失败模式的可执行建议(中文)
- trace_tail: traceback 最后 ~12 行(避免响应体过大)
- context: 调用方标签(如 "intent_init")
"""
exc_type = type(exc).__name__
exc_repr = repr(exc)
# 沿 __cause__ 链找最深异常:httpx 经常把 ConnectError 包成 RemoteProtocolError 等。
root: BaseException = exc
seen: set[int] = {id(root)}
while getattr(root, "__cause__", None) is not None and id(root.__cause__) not in seen:
root = root.__cause__ # type: ignore[assignment]
seen.add(id(root))
code = "unexpected"
hint = "查看 trace_tail 最后一帧定位代码位置;若信息不足请到 backend 日志找完整 traceback。"
if isinstance(root, httpx.ConnectError):
target = str(root).strip().lower()
# loopback 走真 TCP 到 127.0.0.1:<gateway port>;失败 = Gateway 进程没起 / 端口被占。
# outbound 失败(其它 host)= LLM 提供商不可达。
if "127.0.0.1" in target or "localhost" in target or "loopback" in target:
code = "loopback_connect_failed"
hint = (
"本地 Gateway loopback 连接失败:确认后端进程正在监听配置的端口。"
"端口配置优先级(高→低):env DEER_FLOW_GATEWAY_PORT > config.yaml 的 "
"gateway.port > 默认 8001。若部署机改了端口,在 config.yaml 加 "
"`gateway:\\n port: <端口>` 即可。"
)
else:
code = "upstream_connect_failed"
hint = (
"外网/上游服务连接失败:大概率是 LLM 提供商不可达"
"(如 api.deepseek.com 在内网被防火墙拦截)。"
"把 config.yaml 里所有 base_url 改成内网可达的 LLM endpoint,"
"或加白名单/正向代理(HTTPS_PROXY)。"
)
elif isinstance(root, httpx.TimeoutException):
code = "upstream_timeout"
hint = (
"上游请求超时:网络抖动 / 模型推理过慢。"
"先 curl <base_url>/v1/models 确认可达,再看模型是否过载或 max_tokens 过大。"
)
elif isinstance(root, httpx.HTTPStatusError):
code = "upstream_http_error"
status = getattr(getattr(root, "response", None), "status_code", "?")
hint = (
f"上游接口返回 HTTP {status}:可能是 API key 失效、模型名不对、"
"或目标服务自身报错。查看 backend 日志里 response.text 拿详细错误体。"
)
elif isinstance(root, FileNotFoundError):
code = "file_not_found"
hint = (
f"找不到路径:{root}。常见原因:内置 agent 目录(.deer-flow/agents/<id>/)缺失,"
"或 SOUL.md 没拷过来;详见后端开发文档 §11.7。"
)
elif isinstance(root, KeyError):
code = "key_error"
hint = (
f"缺少 key:{root}。若 key 是 agent_name/agent_id 类:大概率内置 agent 目录缺失;"
"若是配置 key:检查 config.yaml schema。"
)
elif isinstance(root, ValueError) and "model" in str(root).lower():
code = "no_models_configured"
hint = "config.yaml 没有任何 models[],或所有条目都解析失败。检查 yaml 语法与 $ENV 引用。"
trace_lines = traceback.format_exception(type(exc), exc, exc.__traceback__)
trace_tail = "".join(trace_lines).splitlines()[-12:]
return {
"code": code,
"message": f"{exc_type}: {exc}" if str(exc) else exc_type,
"exception_type": exc_type,
"exception_repr": exc_repr,
"root_type": type(root).__name__ if root is not exc else None,
"root_repr": repr(root) if root is not exc else None,
"hint": hint,
"trace_tail": trace_tail,
"context": context or None,
}
# ---------------------------------------------------------------------------
# Loopback 辅助函数(Gateway → Gateway,同进程)
#
# 圆桌三路由(intent / recommend / multi_agent)需要从一个 FastAPI handler 里
# 调用本机 Gateway 的另一个 endpoint(`/api/threads`、`/api/agents`、
# `/api/threads/{id}/runs/stream`)。这里用真 TCP loopback 到本机的 127.0.0.1
# + Gateway 监听端口。
#
# ⚠️ 历史踩坑(三个) ⚠️
# 1. 直接走系统代理:开发机若设了 HTTP_PROXY / HTTPS_PROXY,httpx 默认会把
# 127.0.0.1 也走代理 → 上游代理不可达,40s 超时(典型日志 headers_in=42.160s)。
# → 修法:client 用 `trust_env=False`,强制忽略代理 env vars。
#
# 2. 用 request.url.netloc 拼 base URL:Nginx 反代部署时,客户端打到 nginx 的
# public host(如 example.com),nginx 把请求转给 gateway 时 Host header
# 通常是 `gateway:8001` 或 public host;后者 httpx 会真的去 DNS 解析,
# 在 gateway 容器内可能不可达 → 连接失败。
# → 修法:**写死 127.0.0.1**,完全无视 request.url.netloc。
#
# 3. 一度尝试用 httpx.ASGITransport(app=request.app) 进程内直调,看似一劳永逸,
# 但 httpx 0.28 的 ASGITransport 对 streaming response **整段 buffer**,
# 一直收齐所有 body chunks 才一次性把响应给消费方 → SSE 完全失效,前端表现
# 为"调度阶段很长 + 流式结果一次性出现"。已弃用。
# → 详见:git log 6e3a07b(那次改 ASGI 引入)及随后回退。
#
# 端口策略(优先级,从高到低):
# 1. 环境变量 DEER_FLOW_GATEWAY_PORT
# ── 部署侧紧急覆盖通道,适合 docker-compose / k8s 注入或临时调试
# 2. config.yaml 的 `gateway.port`
# ── 部署侧**推荐**的配置方式,改完不需要重启二进制(get_app_config 有 mtime
# 热加载,但仍建议跑一次 reload 以确保所有 worker 拿到新值)
# 3. request.url.port
# ── 浏览器直连 gateway:8001 时拿得到;走 Nginx 80/443 反代时会落到下一档
# 4. 默认 8001(与 `make dev` / `make gateway` / `scripts/start-all.sh` 写死的一致)
#
# 端口在哪里改(给运维 / 部署侧的提示):
# - yaml 配置:打开 config.yaml,加 / 改:
# gateway:
# port: 9999
# - 环境变量:
# export DEER_FLOW_GATEWAY_PORT=9999
#
# 修改这两个函数时务必跑一次 SSE 流式 smoke 验证(见 multi-agent-backend-dev.md
# §13 测试建议),确认 chunks 是逐个到达而不是一次喷出。
# ---------------------------------------------------------------------------
# 写死的最终 fallback。**不要**为了改端口而修改这个常量——优先用 config.yaml /
# env 这两层(它们在部署侧不需要改源码即可生效)。这个值只在 yaml 缺失 + env
# 没设 + 请求里也拿不到合法 port 时才会用到。
_DEFAULT_GATEWAY_PORT = 8001
def _resolve_gateway_port(request: Request) -> int:
"""按四档 fallback 解析 Gateway 监听端口。详见模块顶部「端口策略」。"""
# 第 1 档:环境变量。允许非法值(garbage),只警告并继续下一档,不抛异常。
env_port = os.environ.get("DEER_FLOW_GATEWAY_PORT", "").strip()
if env_port:
try:
port = int(env_port)
if 1 <= port <= 65535:
return port
raise ValueError("out of range")
except ValueError:
logger.warning(
"DEER_FLOW_GATEWAY_PORT=%r 不是合法端口,fallback 到 config.yaml / request / %d",
env_port,
_DEFAULT_GATEWAY_PORT,
)
# 第 2 档:config.yaml 的 gateway.port。GatewayConfig.port 的 pydantic 校验
# 已经保证它在 1-65535 范围内,无需再次校验。读 config 失败时降级,不阻断业务。
try:
configured = get_app_config().gateway.port
if configured:
return configured
except Exception as exc:
# 极端场景:config.yaml 不存在或解析失败。这种情况下 gateway 本身也起不来,
# loopback 端口"猜错"也已经无所谓。仅记 debug 日志方便排查。
logger.debug("read gateway.port from app config failed: %s", exc)
# 第 3 档:从当前请求拿 port。浏览器直连 gateway 时是真实监听端口;
# 走 Nginx 反代时通常是 80/443(public port),那就跳过,落到最终默认。
inbound_port = request.url.port
if inbound_port is not None and inbound_port not in (80, 443):
return inbound_port
# 第 4 档:写死默认。
return _DEFAULT_GATEWAY_PORT
def _loopback_base(request: Request) -> str:
"""返回 `http://127.0.0.1:<port>`,供 loopback 调用拼 URL。
强制 127.0.0.1 是为了避免 Nginx 反代场景下读到经过 Host 改写的 netloc。
端口按 `_resolve_gateway_port` 的四档 fallback 解析。
"""
return f"http://127.0.0.1:{_resolve_gateway_port(request)}"
def _make_loopback_client(request: Request) -> httpx.AsyncClient:
"""构造同进程 loopback 用的 httpx.AsyncClient(真 TCP)。
所有圆桌路由(intent / recommend / multi_agent)的 loopback 调用必须走这里,
保持配置一致。
- `trust_env=False`:**关键**,忽略 HTTP_PROXY / HTTPS_PROXY,防代理拦截 127.0.0.1
- `timeout=None`:SSE 长连接需要,LangGraph runs/stream 可能跑数十秒
- `request` 参数当前未被使用,保留是为了:(a) 未来若需要从 request 取部署
上下文(如多租户端口隔离)可以无侵入扩展;(b) 与调用方现有签名兼容。
"""
return httpx.AsyncClient(
timeout=None,
trust_env=False,
)
def _auth_headers(request: Request) -> dict[str, str]:
token = request.headers.get("authorization")
if not token:
raise HTTPException(status_code=401, detail="Missing Authorization header")
return {
"Authorization": token,
"Content-Type": "application/json",
**internal_run_scope_headers(),
}
async def _create_thread(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
*,
metadata: dict[str, Any] | None = None,
) -> str:
"""``POST /api/threads`` → 新 thread_id(intent/recommend/init 席位均用)。
``metadata`` 写入 thread 元数据,圆桌调用方传 ``_ROUNDTABLE_THREAD_METADATA``
把这些功能型线程标记成可被前端「最近的对话」过滤掉的系统会话。
"""
body: dict[str, Any] = {}
if metadata:
body["metadata"] = metadata
resp = await client.post(f"{base}/api/threads", json=body, headers=headers)
resp.raise_for_status()
return resp.json()["thread_id"]
async def _get_agent(client: httpx.AsyncClient, base: str, headers: dict[str, str], agent_id: str) -> dict[str, Any]:
resp = await client.get(f"{base}/api/agents/{agent_id}", headers=headers)
resp.raise_for_status()
return resp.json()
async def _create_agent(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
*,
name: str,
description: str,
model: str | None,
skills: list[str] | None,
soul: str,
agent_id: str | None = None,
) -> tuple[bool, dict[str, Any] | str]:
payload: dict[str, Any] = {"name": name, "description": description, "soul": soul}
if agent_id is not None:
payload["id"] = agent_id
if model is not None:
payload["model"] = model
if skills is not None:
payload["skills"] = skills
resp = await client.post(f"{base}/api/agents", json=payload, headers=headers)
if resp.status_code == 201:
return True, resp.json()
return False, resp.text
# ---------------------------------------------------------------------------
# 同进程直调 helpers(仅供 init_multi_agent 用,不要扩散到其它地方)
#
# init 阶段串了 N+1 个 thread create + N 个 GET /api/agents + 1 个 POST /api/agents,
# 用 HTTP loopback 串走每次都得 TCP + 鉴权中间件 + Pydantic 反序列化一遍。每个
# 调用 10-30ms,加起来对前端就是「点进 Step 2 后席位灯一颗一颗亮」的卡顿来源。
#
# 这里把这三个调用改成直接吃 app.state 里的 store/checkpointer —— 跑在同一个事
# 件循环里,省掉 TCP + 序列化 + 中间件链。鉴权 / 校验 / 文件副作用按需在直调
# helper 里显式做掉,不要"绕过"任何一项。
#
# 设计边界:这里只覆盖 init 阶段。SSE 长连接(/runs/stream / state)留在 HTTP
# loopback,因为它们本就是别的进程模型(LangGraph 子树),抽出来风险大收益小。
# ---------------------------------------------------------------------------
def _resolve_user_id(request: Request) -> str:
"""与 routers/agents._current_user_id 完全等价,直调路径里用。"""
user = getattr(request.state, "user", None)
if user is not None:
return str(user.id)
return get_effective_user_id()
async def _create_thread_direct(request: Request, *, metadata: dict[str, Any] | None = None) -> str:
"""同进程版的 ``POST /api/threads``。
完全镜像 ``routers/threads.create_thread`` 的写盘动作(thread_meta +
空 checkpoint),但跳过 TCP + Pydantic 反序列化。仅供 init 阶段批量建席位
线程用,失败时抛 HTTPException 由 SSE 帧上报。
``metadata`` 写入 thread 元数据;init 传 ``_ROUNDTABLE_THREAD_METADATA`` 把
席位 / 协调线程标记成系统会话,从前端「最近的对话」里隐藏。
"""
from app.gateway.deps import get_checkpointer, get_thread_store
checkpointer = get_checkpointer(request)
thread_store = get_thread_store(request)
thread_id = str(uuid.uuid4())
now = now_iso()
try:
await thread_store.create(thread_id, assistant_id=None, metadata=metadata or {})
except Exception:
logger.exception("[init-direct] thread_meta create failed for %s", thread_id)
raise HTTPException(status_code=500, detail="Failed to create thread")
config = {"configurable": {"thread_id": thread_id, "checkpoint_ns": ""}}
ckpt_metadata = {
"step": -1,
"source": "input",
"writes": None,
"parents": {},
"created_at": now,
}
async def _cleanup_partial_thread() -> None:
delete_checkpoint = getattr(checkpointer, "adelete_thread", None)
if callable(delete_checkpoint):
try:
await delete_checkpoint(thread_id)
except Exception:
logger.exception("[init-direct] partial checkpoint cleanup failed for %s", thread_id)
try:
await thread_store.delete(thread_id)
except Exception:
logger.exception("[init-direct] partial thread_meta cleanup failed for %s", thread_id)
try:
await checkpointer.aput(config, empty_checkpoint(), ckpt_metadata, {})
except asyncio.CancelledError:
# A sibling seat may fail while this task is between metadata commit and
# checkpoint creation. Compensate before propagating cancellation, under
# shield so a second cancellation cannot abort the cleanup half-way.
await asyncio.shield(_cleanup_partial_thread())
raise
except Exception:
logger.exception("[init-direct] checkpoint create failed for %s", thread_id)
await _cleanup_partial_thread()
raise HTTPException(status_code=500, detail="Failed to create thread")
return thread_id
async def _get_agents_batch_direct(
request: Request,
agent_ids: list[str],
) -> dict[str, dict[str, Any]]:
"""同进程版的 N 个并行 ``GET /api/agents/{id}`` 合并为 1 次 DB 查询。
返回 ``{agent_id: {id, name, description}}``。**严格模式**:任一席位对当前
用户不可见就抛 404 —— 与原 HTTP loopback 的 ``_get_agent`` 行为一致(原路径
单查 404 会 raise,在外层被 except 捕获并 yield error 帧),不要悄悄回退。
"""
from app.gateway.deps import get_agent_store
store = get_agent_store(request)
user_id = _resolve_user_id(request)
records = await store.list_by_ids(agent_ids, user_id)
out: dict[str, dict[str, Any]] = {}
for r in records:
rid = str(r.get("id") or "").lower()
if not rid:
continue
out[rid] = {
"id": rid,
"name": str(r.get("name") or rid),
"description": str(r.get("description") or ""),
}
missing = [sid for sid in agent_ids if sid.lower() not in out]
if missing:
raise HTTPException(
status_code=404,
detail=f"Agent(s) not visible to current user: {', '.join(missing)}",
)
return out
# NOTE: 历史上这里曾有 _create_coordinator_direct(),每次 init 都创建一个带时间戳的
# roundtable-coordinator-<ts36>。2026-05 改造为单例(见 _roundtable_seed.py 与本
# 模块 init_multi_agent 的 docstring),不再需要"每次创建"的直调辅助;改造后由
# ensure_roundtable_functional_agents() 在路由入口落地静态 SOUL,席位列表通过
# _append_thread_message 注入到 coordinator thread 的对话历史中。
async def _append_thread_message(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
thread_id: str,
content: str,
) -> None:
"""向指定 thread 的 checkpoint 追加一条 human 消息(广播上下文,不触发 LLM run)。
失败只打 warning,不阻断主流程——广播是 best-effort,席位真正干活仍靠
前端对该席位 thread 发起的 ``/run/stream``。
走**轻量追加**接口(``POST /messages/append``):只传一条消息,服务端读 checkpoint+追加+写回,
省掉原先 GET /state(下载全量)+POST /state(上传全量)把几万字状态经 loopback 传两遍的开销。
"""
resp = await client.post(
f"{base}/api/threads/{thread_id}/messages/append",
json={"content": content, "type": "human"},
headers=headers,
)
if resp.status_code >= 400:
logger.warning("append message %s failed: %s %s", thread_id, resp.status_code, resp.text[:200])
def _spawn_detached_broadcast(
base: str,
headers: dict[str, str],
message: str,
agent_threads: dict[str, str],
skip: list[str],
) -> None:
"""后台 fire-and-forget 广播,**用独立 client**,即使发起它的请求 SSE 已结束也能跑完。
广播是 best-effort 的「上下文播报」(告知其它席位某人交付了什么),原本在总控 yield 派活
帧**之前**被 ``await``,导致每个席位都要等几十秒(广播要读写各席位的巨大 thread 状态)
才开始执行 —— 实测 ``dispatch_broadcasts_await`` 飙到 27s。改成 detached 后台任务后,总控
立即 yield 派活、前端立即启动席位,广播在后台慢慢落地,彻底移出关键路径。失败只记日志。
"""
async def _run() -> None:
try:
async with httpx.AsyncClient(timeout=None, trust_env=False) as bg:
await _broadcast(bg, base, headers, message, agent_threads, skip)
except Exception as exc: # noqa: BLE001
logger.warning("detached broadcast failed: %s", exc)
# create_task 后不持有引用:任务挂在事件循环上,独立于本请求生命周期跑完。
asyncio.ensure_future(_run())
async def _broadcast(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
message: str,
agent_threads: dict[str, str],
skip: list[str],
) -> None:
"""并行向 ``agent_threads`` 中除 ``skip`` 外的所有 thread 追加同一条 human 消息。"""
targets = [tid for name, tid in agent_threads.items() if name not in skip]
if not targets:
return
await asyncio.gather(
*(_append_thread_message(client, base, headers, tid, message) for tid in targets),
return_exceptions=True,
)
_TOOL_CALL_PATTERN = re.compile(r'\.py\s+"([^"]*)"\s+"([^"]*)"')
# 子智能体交付的「后广播」里,正文广播给**其它席位**时的截断上限。
# 背景:席位交付动辄几万字;原实现把整篇交付广播进**每一个**其它席位 thread,导致
# 每个 thread 状态滚雪球,之后每次状态读/写(广播、交付状态核对)都要搬运巨量数据,
# 延迟复利式恶化(实测 dispatch_broadcasts_await 飙到 30s)。其它席位只需"知道某席位
# 交付了什么"的摘要即可;**完整交付仍原样发给总控 thread**(总控做最终综合要看全文),
# 也仍通过 SSE 完整给到前端。仅截断"广播给兄弟席位"这一路。
_BROADCAST_CONTENT_MAX_CHARS = 1500
def _truncate_broadcast_content(prefix: str, content: str) -> str:
"""构造广播给**兄弟席位**的交付摘要消息:超长正文截断到上限并标注省略。
``content`` 未超限时原样拼接(等价于完整版),超限时截断 + 追加省略说明。
总控那一路不走本函数,始终用完整正文(见 ``_special_run`` 的 ``full_msg``)。
"""
if len(content) <= _BROADCAST_CONTENT_MAX_CHARS:
return prefix + content
return (
prefix
+ content[:_BROADCAST_CONTENT_MAX_CHARS]
+ f"\n…(完整交付约 {len(content)} 字,已省略;以该席位自身交付为准)"
)
# Mirrors the upstream agent-id validator in the LangGraph runtime. Reject early
# so callers get a clear 400 here instead of a stream that aborts with HTTP 422
# from the loopback /api/threads/.../runs/stream endpoint.
_AGENT_ID_PATTERN = re.compile(r"^[A-Za-z0-9_-]+$")
def _validate_agent_id(value: str, field: str) -> str:
if not _AGENT_ID_PATTERN.match(value):
raise HTTPException(
status_code=400,
detail=(
f"{field} '{value}' is not a valid agent id. "
f"Must match ^[A-Za-z0-9_-]+$ (use the hyphen-case id, not the display name)."
),
)
return value
def _extract_dispatch(tool_call: dict[str, Any]) -> tuple[str, str]:
"""Return ``(agent_name, task)`` parsed from a leader tool call.
Two formats are accepted:
1. **Structured** — ``{"name": "agent_orchestration", "args": {"agent_name": ..., "task": ...}}``.
This is what most LLMs emit when told to "use the agent_orchestration skill",
and the form the roundtable coordinator's SOUL prompt elicits.
2. **Legacy bash argv** — ``{"name": "bash", "args": {"command": '... .py "X" "Y" ...'}}``.
Kept as a fallback so older skills that dispatch through bash still work.
"""
name = (tool_call.get("name") or "").lower()
args = tool_call.get("args")
# Format 1: structured agent_orchestration call.
if name == "agent_orchestration" and isinstance(args, dict):
agent_name = str(args.get("agent_name") or "").strip()
task = str(args.get("task") or "").strip()
if agent_name and task:
return _strip_display_suffix(agent_name), task
# Format 2: legacy bash + .py "X" "Y".
match = _TOOL_CALL_PATTERN.search(str(args))
if match:
return _strip_display_suffix(match.group(1)), match.group(2)
return "", ""
def _strip_display_suffix(agent_name: str) -> str:
"""Tolerate models that paste the human-readable suffix into agent_name.
The coordinator SOUL lists each seat as ``agent_name: roundtable-x(中文):...``
so weaker models sometimes echo the entire ``roundtable-x(中文)`` label
back as the dispatch target. Strip everything from the first non-id
character onward so we recover the bare hyphen-case id; if what's left
still violates the id pattern we hand it back untouched and let the
upstream validator reject it.
"""
head = re.split(r"[^A-Za-z0-9_-]", agent_name, maxsplit=1)[0]
return head or agent_name
def _flatten_content(content: Any) -> str:
if isinstance(content, str):
return content
if isinstance(content, list):
parts: list[str] = []
for item in content:
if isinstance(item, dict):
parts.append(str(item.get("text", "")))
else:
parts.append(str(item))
return "".join(parts)
return str(content)
def _last_ai_raw_text(messages: list[Any]) -> str:
"""最后一条有原文的 AI 消息(含思考标签),用于区分 thinking_only 与 empty。"""
for msg in reversed(messages or []):
if not isinstance(msg, dict):
continue
if (msg.get("type") or "").lower() not in ("ai", "aimessage", "aimessagechunk"):
continue
raw = _flatten_content(msg.get("content", ""))
if raw.strip():
return raw
return ""
def _visible_delivery_from_messages(messages: list[Any]) -> str:
"""本轮可见正文:最后一条 human 之后、剥掉思考仍非空的 AI 文本。"""
last_human = -1
for i, msg in enumerate(messages or []):
if isinstance(msg, dict) and (msg.get("type") or "").lower() in ("human", "user"):
last_human = i
for msg in reversed((messages or [])[last_human + 1 :]):
if not isinstance(msg, dict):
continue
if (msg.get("type") or "").lower() not in ("ai", "aimessage", "aimessagechunk"):
continue
visible = visible_seat_text(_flatten_content(msg.get("content", "")))
if visible:
return visible
return ""
def _run_payload(
*,
agent_name: str,
model_name: str,
thread_id: str,
new_message: str,
excluded_tools: list[str],
skill_stop_names: list[str],
thinking_enabled: bool = True,
reasoning_effort: str = "medium",
subagent_enabled: bool = False,
stream_custom: bool = False,
multitask_strategy: str = "reject",
extra_context: dict[str, Any] | None = None,
files: list[dict[str, Any]] | None = None,
) -> dict[str, Any]:
"""构造 ``POST /api/threads/{thread_id}/runs/stream`` 的请求体。
- ``context.agent_name``:本次 run 使用的 agent(内置 id 或 init 创建的 coordinator id)。
- ``excluded_tools``:按路由裁剪工具面(intent 仅澄清、special 禁派活等)。
- ``skill_stop_names``:leader 侧配合 SkillStopMiddleware 在 ``agent_orchestration``
调用后截断图,以便网关从 AIMessage.tool_calls 解析派活列表。
- ``thinking_enabled`` / ``reasoning_effort``:默认开 CoT + medium,适合 Step 1 澄清
与子智能体交付这类需要推理的场景。**推荐弹窗与总控派活**是短决策任务,调用方
应传 ``thinking_enabled=False`` + ``reasoning_effort="low"``,省掉模型先吐
reasoning tokens 才开口的时间,显著降低前端"调度中"首字节延迟。
"""
# 用户上传的文件元数据透传给 UploadsMiddleware(与普通聊天同链路):
# before_agent 钩子读 additional_kwargs.files,生成 <uploaded_files> 清单注入消息,
# 之后席位/总控用内置 read_file / grep / view_image 工具按需读取。空时落空 dict,行为不变。
additional_kwargs: dict[str, Any] = {}
if files:
additional_kwargs["files"] = files
payload: dict[str, Any] = {
"input": {
"messages": [
{
"type": "human",
"content": [{"type": "text", "text": new_message}],
"additional_kwargs": additional_kwargs,
}
]
},
"metadata": {"run_scope": RUN_SCOPE_ORCHESTRATION_CHILD},
"config": {"recursion_limit": 1000},
"context": {
"agent_name": agent_name,
"model_name": model_name,
"mode": "pro",
"reasoning_effort": reasoning_effort,
"thinking_enabled": thinking_enabled,
"is_plan_mode": False,
# 默认 False;圆桌「多智能体问答(ultra)」档会让席位 _special_run 传 True,
# 给席位解锁 task 子代理工具(席位 policy 未禁 task,故可用)。
"subagent_enabled": subagent_enabled,
"thread_id": thread_id,
# 圆桌四个 router(intent / recommend / leader / special)都是临时任务,
# 用户个人记忆与本次研讨无关。MemoryMiddleware 的 Hindsight recall 是
# 同步阻塞 HTTP 调用(单次 timeout 15s),如果记忆服务慢/不通会让前端
# "调度中"卡很久;这里默认跳过。
"memory_recall_disabled": True,
# ⚠️ 同时关掉 builtin 记忆**注入**:圆桌四个角色(intent / recommend /
# leader / special)都是固定单例 agent,其 memory.json 跨所有会话累积。
# 总控尤其危险——上一个任务里它臆造的席位名(如 data-specialist)会被存进
# memory.json,下一个新任务又被注入系统提示而召回 → 派给根本没初始化的
# 席位 → run/stream 报 400 "agent_name ... missing from d_agent_thread_id"。
# 圆桌是一次性研讨,本就不该吃历史会话记忆,这里统一关注入根治该污染。
"memory_injection_enabled": False,
},
# ultra 席位会派子智能体(task 工具),其 task_started/running/completed 进度走
# **custom** 流(StreamWriter)。仅这种场景加 "custom",让前端任务卡片能显示子智能体
# 实时进度;其余角色不加,避免无谓帧。
"stream_mode": (
["messages-tuple", "values", "custom"] if stream_custom else ["messages-tuple", "values"]
),
"stream_subgraphs": True,
"stream_resumable": True,
"assistant_id": "lead_agent",
"on_disconnect": "continue",
# ⚠️ 圆桌 run 必须能被「下一次同 thread 的 run」抢占。圆桌每个 thread 同一时刻
# 只应有一个活跃 run,但 on_disconnect="continue" 让前端 abort SSE 后**后端 run
# 仍在跑**(为可恢复);此时用户「人工干预」立刻发起新 leader run 打到同一总控
# thread,默认 reject 策略就会 409「already has an active run」→ 被 _stream_upstream
# 包成 502。调用方(leader / special)传 multitask_strategy="rollback",让新 run 先
# 取消该 thread 上没结束的旧 run 再继续 —— 这正是「最新一轮/干预指令优先」的语义。
"multitask_strategy": multitask_strategy,
"excluded_tools": excluded_tools,
"skill_stop_names": skill_stop_names,
}
# 额外注入 run context(如圆桌席位的 roundtable_peer_deliveries:前序席位完整交付,
# 供 read_peer_delivery 工具按需读取)。只进 context dict,不进 prompt → 不调工具就不耗 token。
if extra_context:
payload["context"].update(extra_context)
return payload
# ---------------------------------------------------------------------------
# POST /api/multi-agent/init — Step 2 席位 + 总控初始化(SSE)
# ---------------------------------------------------------------------------
@router.post("/init")
async def init_multi_agent(request: Request, payload: dict = Body(...)) -> StreamingResponse:
"""流式初始化圆桌 Step 2 所需的 thread 映射与协调智能体。
请求体::
{
"agent_names": ["roundtable-intelligence", ...], // 用户选的席位 id
"main_agent_name": "<可选,已废弃>", // 历史兼容字段,新代码忽略,统一用 COORDINATOR_AGENT_ID
"model": "<可选>" // 可选;不传由 lead_agent 回退到 config.yaml models[0]
}
SSE 事件(每条 ``data: {...}\\n\\n``)::
seat_ready — 某一席位 thread 已创建(``asyncio.as_completed`` 顺序,先完先发)
coordinator_ready — 总控 thread 就绪 + 已注入本次席位列表
init_done — ``thread_ids`` 全量映射,前端存 ``threadIds`` 并开始编排
error — 失败详情(同步校验失败则直接 HTTP 4xx,不进 SSE)
协调智能体改造说明
----------------
协调智能体现在是 **单例**(id 固定为 ``roundtable-coordinator``,SOUL 见
``_roundtable_seed.COORDINATOR_AGENT_SOUL``),由 ``ensure_roundtable_functional_agents()``
在首次调用时自动落地。每次 init **不再创建新的 coordinator agent**(避免
``.deer-flow/agents/`` 累积 ``roundtable-coordinator-<ts36>`` 残留)。
本次圆桌的「可调度席位列表」由 init 在创建好 coordinator thread 之后,以一条
human 消息追加到 thread 历史里,leader run 每轮都能从对话历史里读到。这样
SOUL 长期稳定,席位列表随会话动态注入。前端为兼容旧 client 仍可传
``main_agent_name`` 字段,但后端会忽略并使用 ``COORDINATOR_AGENT_ID``。
"""
# 调用前自检:与 intent/recommend 一样,缺失时就地补建,避免 /run/stream 报 500。
ensure_roundtable_functional_agents()
agent_names_in = payload.get("agent_names") or []
agent_names: list[str] = [str(name).lower() for name in agent_names_in]
model: str | None = payload.get("model")
# main_agent_name 是历史字段,2026-05 改造后忽略,统一用单例 id。保留接收但不再用前端值,
# 避免下游再次创建带时间戳的临时 coordinator。
main_agent_name = COORDINATOR_AGENT_ID
if not agent_names:
raise HTTPException(status_code=400, detail="agent_names is required")
for name in agent_names:
_validate_agent_id(name, "agent_names entry")
def _frame(payload: dict[str, Any]) -> str:
return f"data: {json.dumps(payload, ensure_ascii=False)}\n\n"
async def stream_generator() -> AsyncIterator[str]:
# init 阶段 fast path:全部走同进程直调(_create_thread_direct /
# _get_agents_batch_direct / _create_coordinator_direct),不再走 HTTP
# loopback —— TCP + 鉴权中间件 + Pydantic 反序列化每次 10-30ms,N+1 个席位
# 累计就是"点进 Step 2 后席位一颗颗亮"的卡顿来源。
try:
t_init_start = time.perf_counter()
# 1) 一次 DB 查询拿到所有席位的 name/description,替代之前的 N 个并行
# GET /api/agents/{id}。SOUL 拼接需要描述,这里一次性拿齐。
agent_meta = await _get_agents_batch_direct(request, agent_names)
t_meta_done = time.perf_counter()
# 2) 每个席位 thread 并行创建(直调 thread_store + checkpointer),
# 用 asyncio.as_completed 保持"先完先发 seat_ready"的体验。
async def _prepare_sub_agent(sub_id: str) -> tuple[str, str, str, str]:
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
meta = agent_meta.get(sub_id.lower(), {})
return (
sub_id.lower(),
thread_id,
str(meta.get("name") or sub_id),
str(meta.get("description") or ""),
)
tasks = [asyncio.create_task(_prepare_sub_agent(sid)) for sid in agent_names]
prepared: list[tuple[str, str, str, str]] = []
try:
for fut in asyncio.as_completed(tasks):
sid, thread_id, display_name, description = await fut
prepared.append((sid, thread_id, display_name, description))
yield _frame({
"event": "seat_ready",
"agent_name": sid,
"thread_id": thread_id,
"display_name": display_name,
"description": description,
})
except Exception:
# Cancel any still-pending sibling so partially-prepared seats
# don't keep running past the stream close.
for t in tasks:
if not t.done():
t.cancel()
raise
t_seats_done = time.perf_counter()
agent_threads: dict[str, str] = {sid: thread_id for sid, thread_id, _, _ in prepared}
seat_lines: list[str] = [
f"- agent_name: {sid}({display_name}):{description}"
for sid, _, display_name, description in prepared
]
# 协调智能体改造为**单例**(id = roundtable-coordinator):
# - SOUL 已经由 ensure_roundtable_functional_agents() 在路由入口落地,
# 通用协调规则常驻磁盘,无需每次重写。
# - 本次"可调度席位清单"以 human 消息形式 append 到 coordinator
# thread 历史里,leader run 每轮都能在对话历史最前面看到。
# 这样席位列表随会话动态变化,但磁盘上不再累积 roundtable-coordinator-<ts36>。
#
# 注意:model 字段在 lead_agent run 时通过 context.model_name 注入,
# 不需要写回到 agent config.yaml(否则会污染单例的配置)。
seat_introduction = (
"【圆桌系统初始化】本次研讨可调度的席位清单如下,请严格按 agent_name "
"通过 agent_orchestration 技能派活,不要派给不在清单内的 agent:\n"
+ "\n".join(seat_lines)
)
coord_thread = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
agent_threads[main_agent_name] = coord_thread
# 把席位列表注入 coordinator thread 的对话历史。HTTP loopback 调用
# 走 /api/threads/{tid}/state(读+写),开销 ~10-30ms,在 init 整体
# 几百 ms 的尺度上微不足道,但比起手撕 langgraph checkpoint 内部结构
# 风险低、与 _broadcast / _append_thread_message 一套语义,长期可维护。
try:
base = _loopback_base(request)
headers = _auth_headers(request)
async with _make_loopback_client(request) as seed_client:
await _append_thread_message(
seed_client,
base,
headers,
coord_thread,
seat_introduction,
)
except Exception:
# 注入失败不阻断 init:leader run 仍能跑,只是看不到席位清单,
# 可能派活给错的 id。会被 _strip_display_suffix / _validate_agent_id
# 兜底,前端能看到错误帧;运维通过 backend 日志找到根因再修复。
logger.exception("[init-direct] failed to seed coordinator thread with seat list")
t_coord_done = time.perf_counter()
logger.info(
"[init-direct] meta=%.3fs seats=%.3fs coord=%.3fs total=%.3fs seats_count=%d",
t_meta_done - t_init_start,
t_seats_done - t_meta_done,
t_coord_done - t_seats_done,
t_coord_done - t_init_start,
len(prepared),
)
yield _frame({
"event": "coordinator_ready",
"main_agent_name": main_agent_name,
"thread_id": coord_thread,
})
yield _frame({
"event": "init_done",
"main_agent_name": main_agent_name,
"thread_ids": agent_threads,
})
except HTTPException as exc:
logger.warning("multi-agent init HTTPException: %s %s", exc.status_code, exc.detail)
await record_foreground_diag(
request, stage="step2_init", level="error", event="init_http_error",
message=f"会商初始化失败 [{exc.status_code}]:{exc.detail}",
detail={"status_code": exc.status_code, "detail": str(exc.detail)},
)
yield _frame({"event": "error", "detail": f"[{exc.status_code}] {exc.detail}"})
except Exception as exc:
logger.exception("multi-agent init unexpected failure")
await record_foreground_diag(
request, stage="step2_init", level="error", event="init_unexpected",
message=f"会商初始化未预期异常:{exc!r}",
detail={"type": type(exc).__name__, "error": str(exc)},
)
yield _frame({"event": "error", "detail": f"unexpected: {exc!r}"})
return StreamingResponse(
stream_generator(),
media_type="text/event-stream",
headers={
"Cache-Control": "no-cache",
"X-Accel-Buffering": "no",
"Connection": "keep-alive",
},
)
@router.post("/report/init")
async def init_report_thread(request: Request) -> dict[str, str]:
"""为 Step 3「结果绘制」创建一个供方案可视化总结智能体(``roundtable-report``)
运行的线程,返回 ``{"agent_name", "thread_id"}``。
设计要点:
- ``roundtable-report`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进
DB),这里**只建一个轻量 thread**,绝不重建 agent;
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
``agent_name="roundtable-report"``、``d_agent_thread_id={REPORT_AGENT_ID: thread_id}``)
跑这一轮,智能体把图表 HTML 写进该线程的 ``outputs/``,再由沙箱预览。
"""
try:
ensure_roundtable_functional_agents()
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
except Exception as exc:
await record_foreground_diag(
request, stage="step3_report", level="error", event="thread_init_failed",
message=f"创建结果绘制线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
agent_id=REPORT_AGENT_ID,
)
raise
return {"agent_name": REPORT_AGENT_ID, "thread_id": thread_id}
@router.post("/summary/init")
async def init_summary_thread(request: Request) -> dict[str, str]:
"""为 Step 3「总结报告」创建一个供方案总结报告智能体(``roundtable-summary``)
运行的线程,返回 ``{"agent_name", "thread_id"}``。
与 ``/report/init`` 同构,只是换成另一个内置单例 ``roundtable-summary``:
- ``roundtable-summary`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进
DB),这里**只建一个轻量 thread**,绝不重建 agent;
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
``agent_name="roundtable-summary"``、``d_agent_thread_id={SUMMARY_AGENT_ID: thread_id}``)
跑这一轮(后续问答/增量改**复用同一线程**,不重新 init),智能体把 Markdown 报告
写进该线程的 ``outputs/方案总结报告.md``,再由沙箱预览。
"""
try:
ensure_roundtable_functional_agents()
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
except Exception as exc:
await record_foreground_diag(
request, stage="step3_summary", level="error", event="thread_init_failed",
message=f"创建总结报告线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
agent_id=SUMMARY_AGENT_ID,
)
raise
return {"agent_name": SUMMARY_AGENT_ID, "thread_id": thread_id}
@router.post("/position-action-plan/init")
async def init_position_action_plan_thread(request: Request) -> dict[str, str]:
"""Create the reusable thread for the position-roundtable action planner.
The planner is a built-in singleton agent just like ``roundtable-summary``;
only its thread is per session. The position workspace persists this id
after the first successful run so follow-up revisions retain the original
plan context.
"""
try:
ensure_roundtable_functional_agents()
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
except Exception as exc:
await record_foreground_diag(
request,
stage="position_action_plan",
level="error",
event="thread_init_failed",
message=f"创建岗位行动规划线程失败:{exc}",
detail={"type": type(exc).__name__, "error": str(exc)},
agent_id=ACTION_PLAN_AGENT_ID,
)
raise
return {"agent_name": ACTION_PLAN_AGENT_ID, "thread_id": thread_id}
@router.post("/dashboard/init")
async def init_dashboard_thread(request: Request) -> dict[str, str]:
"""为 Step 3「大屏多页」创建一个供大屏多页数据智能体(``roundtable-dashboard``)
运行的线程,返回 ``{"agent_name", "thread_id"}``。
与 ``/report/init`` / ``/summary/init`` 同构,换成第三个内置单例
``roundtable-dashboard``(专职「逐席位分析 → report-json」,run policy 为 ``dashboard``,
**禁用一切工具**):
- ``roundtable-dashboard`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进 DB),
这里**只建一个轻量 thread**,绝不重建 agent;
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
``agent_name="roundtable-dashboard"``、``d_agent_thread_id={DASHBOARD_AGENT_ID: thread_id}``)
跑这一轮(后续问答/增量改**复用同一线程**,不重新 init),智能体在对话里输出
```report-json,前端用 ``extractReportData`` 抽取后按当前主题渲染大屏。
"""
try:
ensure_roundtable_functional_agents()
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
except Exception as exc:
await record_foreground_diag(
request, stage="step3_dashboard", level="error", event="thread_init_failed",
message=f"创建大屏数据线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
agent_id=DASHBOARD_AGENT_ID,
)
raise
return {"agent_name": DASHBOARD_AGENT_ID, "thread_id": thread_id}
@router.post("/structure/init")
async def init_structure_thread(request: Request) -> dict[str, str]:
"""为 Step 3「入库」创建一个供结构化抽取智能体(``roundtable-structure``)运行的线程,
返回 ``{"agent_name", "thread_id"}``。
与 ``/report/init`` / ``/summary/init`` / ``/dashboard/init`` 同构,换成内置单例
``roundtable-structure``。区别在于本轮**不是抽流程图**,而是让它**真正调用入库技能**
(run policy 为 ``ingest``——放开 ``read_file`` / ``bash`` / ``write_file`` 让技能能执行):
- ``roundtable-structure`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进 DB),
这里**只建一个轻量 thread**,绝不重建 agent;
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
``agent_name="roundtable-structure"``、``d_agent_thread_id={STRUCTURE_AGENT_ID: thread_id}``)
跑这一轮,把 taskId + 总结报告 + 结构化流程图交给它,由它挑选合适的入库技能(名称
可能含「六步法」/自行判断适用性)读 SKILL.md 并按其契约调用完成入库。
"""
try:
ensure_roundtable_functional_agents()
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
except Exception as exc:
await record_foreground_diag(
request, stage="step3_structure", level="error", event="thread_init_failed",
message=f"创建数据入库线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
agent_id=STRUCTURE_AGENT_ID,
)
raise
return {"agent_name": STRUCTURE_AGENT_ID, "thread_id": thread_id}
# ---------------------------------------------------------------------------
# 上游 LangGraph SSE 透传(intent / recommend / leader / special 共用)
# ---------------------------------------------------------------------------
# 回环 /runs/stream 在**响应头阶段**返回这些状态码时,视为「单 worker + 单 sqlite 锁被
# 并发打满」造成的瞬时失败,退避重试(详见 _stream_upstream 内注释):
# 429 限流;500/503/504 高并发下 start_run 里 DB/checkpoint 写抖动。
# 不含 400/401/403/404/422 等确定性错误——那些重试也没用,立即上报。
#
# ⚠️ 409 **不在**重试集合里。409 = 同 thread 已有活跃 run(``multitask_strategy=reject``
# 拒绝)。此时重发**不是**幂等的:LangGraph 会把同一条 human message 再追加一次 → 用户
# 这一轮被处理两遍(重复澄清卡 / 双发意图),或两次退避后仍 502 —— 这正是用户报的
# 「确定意图接口报错」。故 409 必须透传:由上层(intent_turns 去重 / 前端按 client_turn_id
# 等待)决定是等待还是放弃,而不是盲目重发。其余(rollback)调用方本就靠抢占取消在跑 run,
# 几乎不触 409,移除亦无碍。
_UPSTREAM_RETRYABLE_STATUSES = frozenset({429, 500, 502, 503, 504})
# 重试次数(不含首次)。总尝试 = 1 + _UPSTREAM_MAX_RETRIES。
_UPSTREAM_MAX_RETRIES = 2
# 指数退避基数(秒):0.4 → 0.8 → …,足够让在跑的席位 run 让出 sqlite 锁,又不至于明显拖慢。
_UPSTREAM_RETRY_BASE_DELAY = 0.4
async def _stream_upstream(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
thread_id: str,
body: dict[str, Any],
) -> AsyncIterator[tuple[str | None, list[str] | None]]:
"""消费 ``/api/threads/{thread_id}/runs/stream``,逐行 yield 原始 SSE。
产出元组:
- ``(line, None)`` — 可原样转发给浏览器(含 ``event:`` / ``data:`` 行);
- ``(None, captured)`` — 流结束,``captured`` 为所有 ``data:`` 行文本列表,
供 ``_parse_last_messages`` 从倒数 values 快照提取 ``messages``。
日志 ``[upstream-timing]``:排查首 token 慢;headers_in 长尾若 >> 网络往返(loopback
本机往返通常 <10ms),怀疑 HTTP_PROXY 拦了回环——检查 client 是否走 _make_loopback_client
(trust_env=False)。
"""
url = f"{base}/api/threads/{thread_id}/runs/stream"
captured: list[str] = []
agent_for_log = (body.get("context") or {}).get("agent_name") or thread_id
# 头阶段非 200 的瞬时重试。
# 背景:网关是**单 worker** + checkpointer 是**单连接 aiosqlite**(全进程一把
# asyncio.Lock 串行 checkpoint 读写)。圆桌第三步进入时,前端把「方案总结/结果绘制/
# 大屏」三个重 run **同一刻并行**打出来,叠加第二步 on_disconnect=continue 仍在后台
# 跑的席位 run,瞬时把单锁/单 worker 打满 —— 此时回环子请求 /runs/stream 的 start_run
# 里那串 DB / checkpoint 写有概率抛错(或同 thread 抢占竞态),返回非 200,被本函数原样
# 包成 502 漏给前端(用户侧表现为"进第三步概率性 502")。
# 关键安全前提:这个非 200 发生在**响应头阶段**,本函数**尚未 yield 任何 SSE 行**给下游,
# 所以"换条连接重发"是幂等安全的(不会重复透传半截流)。对可恢复的瞬时状态码退避重试几次,
# 把绝大多数并发尖峰造成的偶发失败吃掉;非可恢复状态码(如 400/401/403/404)立即按原逻辑
# 报错,不浪费重试。一旦进入流式(已 yield 行)就绝不重试。
last_status = 0
last_body = ""
for attempt in range(_UPSTREAM_MAX_RETRIES + 1):
t_send = time.perf_counter()
async with client.stream("POST", url, json=body, headers=headers) as resp:
t_headers = time.perf_counter()
if resp.status_code != 200:
last_status = resp.status_code
last_body = (await resp.aread()).decode("utf-8", "replace")[:1000]
if resp.status_code in _UPSTREAM_RETRYABLE_STATUSES and attempt < _UPSTREAM_MAX_RETRIES:
delay = _UPSTREAM_RETRY_BASE_DELAY * (2**attempt)
logger.warning(
"[upstream-retry] %s 回环 %s 返回 HTTP %s,%.2fs 后重试(第 %d/%d 次):%s",
agent_for_log,
url,
resp.status_code,
delay,
attempt + 1,
_UPSTREAM_MAX_RETRIES,
last_body[:200],
)
await asyncio.sleep(delay)
continue # 重连:换条新连接重发同一 body(尚未 yield 任何行,幂等安全)
raise HTTPException(
status_code=502,
detail=f"upstream {url} returned HTTP {resp.status_code}: {last_body}",
)
logger.info(
"[upstream-timing] %s headers_in=%.3fs%s",
agent_for_log,
t_headers - t_send,
f" (retried x{attempt})" if attempt else "",
)
first_line = True
async for line in resp.aiter_lines():
if not line:
continue
if first_line:
first_line = False
logger.info(
"[upstream-timing] %s first_line=%.3fs (LangGraph runtime startup + LLM first byte)",
agent_for_log,
time.perf_counter() - t_headers,
)
yield line, None
if line.startswith("data:"):
captured.append(line)
yield None, captured
return
# 理论上不可达(循环要么 return、要么 raise);留一条兜底防御。
raise HTTPException(
status_code=502,
detail=f"upstream {url} returned HTTP {last_status}: {last_body}",
)
def _parse_last_messages(captured: list[str]) -> list[dict[str, Any]] | None:
"""Return the full ``messages`` list from the most recent ``values`` snapshot.
The upstream sends the final ``values`` frame second-to-last (the very last
line is the ``end`` event with ``null``). We tolerate trailing non-JSON
frames (e.g. ``data: null``) by scanning backwards for the first frame that
successfully parses as a state object with a ``messages`` key.
Returns ``None`` when the stream carried **no parseable ``values`` frame**
(empty capture, or only ``data: null`` / token deltas). 这通常**不是真正的
上游错误**,而是该 run 被 ``multitask_strategy=rollback`` 抢占 / 客户端断开 /
中途 abort —— 流被截断,没来得及发出最终 values 快照。调用方(leader/special)据此
**静默收尾**,绝不能把它当 502 抛给前端:抢占场景下这条 SSE 多半已被前端
``abortAllInFlight`` 放弃,硬报 502 只会在 UI 上凭空冒出一条错误。
"""
if not captured:
return None
for raw_line in reversed(captured):
raw = raw_line[5:].strip() if raw_line.startswith("data:") else raw_line.strip()
if not raw or raw == "null":
continue
try:
parsed = json.loads(raw)
except json.JSONDecodeError:
continue
if isinstance(parsed, dict) and "messages" in parsed:
messages = parsed.get("messages") or []
if isinstance(messages, list):
return messages
return None
def _find_trailing_ai_text(messages: list[dict[str, Any]], *, exclude_text: str = "") -> str:
"""Return the text content of the latest AIMessage with no tool_calls.
Used by the clarification branch: ClarificationMiddleware ends the run
after emitting the structured ToolMessage, but the model often produces a
richer human-readable follow-up AIMessage right before the interrupt.
Prefer that text over the brief preamble that accompanied the
ask_clarification tool_call. When the only trailing AIMessage matches the
preamble (``exclude_text``), return empty so the caller can fall back to
the question itself.
"""
for msg in reversed(messages):
if not isinstance(msg, dict):
continue
if (msg.get("type") or "").lower() not in ("ai", "aimessage", "aimessagechunk"):
continue
if msg.get("tool_calls"):
continue
text = _flatten_content(msg.get("content", "")).strip()
if not text:
continue
if exclude_text and text == exclude_text.strip():
continue
return text
return ""
def _find_dispatch_message(messages: list[dict[str, Any]]) -> dict[str, Any]:
"""Locate the latest AIMessage from THIS turn.
Trailing ToolMessages (skill_stop / clarification ack injected by
``SkillStopMiddleware`` or ``ClarificationMiddleware``) are skipped because
the actionable tool_calls live on the AIMessage that emitted them. We stop
at the **first AIMessage encountered** — its ``tool_calls`` (or lack
thereof) is the authoritative signal for this turn.
历史踩坑:之前实现是"一直往前翻直到撞到带 tool_calls 的 AI",看似优雅,
实际上让 leader 永远收敛不了:本轮 LLM 不论怎么共识,只要历史上派过活,
上一轮的派活 AI 就会被翻出来再用一次,前端会把同样的 task 又发给子智能体,
陷入死循环。现在停在最新一条 AI 上,把判定交给调用方:`tool_calls=[]` →
共识分支,有 tool_calls → 派活分支。
"""
for msg in reversed(messages):
if not isinstance(msg, dict):
continue
msg_type = (msg.get("type") or "").lower()
if msg_type in ("tool", "toolmessage"):
continue # 跳过末尾 skill_stop / clarification ack
if msg_type in ("ai", "aimessage", "aimessagechunk"):
return msg
break # 撞到 Human/其他类型,本轮没 AI(异常路径,留空让上游报错)
return {}
def _partition_dispatch(
tool_calls: list[dict[str, Any]],
valid_seat_ids: set[str],
) -> tuple[list[tuple[str, str]], list[str]]:
"""把总控的 agent_orchestration 派活拆成「合法 / 非法」两份。
返回 ``(valid, invalid)``:
- ``valid`` —— ``[(seat_id, task), ...]``,``seat_id`` 确实是本轮已初始化的席位;
- ``invalid`` —— ``[seat_id, ...]``,模型臆造 / 不在本轮 roster 内的 id。
弱模型(deepseek-chat)会把 SOUL 里的示例 hex id 当真照抄、或臆造同款 32 位 hex,
这些 id 不在 ``agent_threads`` 里;若不拦截会漏给前端,到 ``/run/stream`` 才报 400
把整个 Step 2 打挂。校验放在派活解析处,只丢非法、保留同轮合法席位。
"""
valid: list[tuple[str, str]] = []
invalid: list[str] = []
for tool_call in tool_calls:
sub_name, task_text = _extract_dispatch(tool_call)
if not sub_name:
continue
if sub_name in valid_seat_ids:
valid.append((sub_name, task_text))
else:
invalid.append(sub_name)
return valid, invalid
def _build_dispatch_correction(invalid_names: list[str], valid_seat_ids: set[str]) -> str:
"""整轮派活全非法时,回灌给总控 thread 的系统纠正消息文本。
列出真实席位 id,要求模型**逐字符复制**重派,杜绝「凭示例/记忆构造 id」。
"""
bad = "、".join(invalid_names)
roster = "、".join(sorted(valid_seat_ids))
return (
"【系统纠正】你刚才通过 agent_orchestration 派活的 agent_name("
f"{bad})不在本次圆桌的席位清单里,无法执行。请**只**从下面这份真实席位清单里"
"逐字符复制 agent_name 重新派活,不要凭记忆、示例或猜测自行构造 id:\n"
f"{roster}"
)
def _build_seat_roster_block(seats: list[tuple[str, dict[str, Any]]]) -> str:
"""构造注入到总控**本轮输入最前面**的「真实席位清单」文本。
``seats`` 为 ``[(seat_id, {name, description}), ...]``。
为什么要每轮把清单塞进输入,而不是只靠 init 时 _append_thread_message 注入历史:
圆桌席位清单原本只在 init 时以一条 human 消息异步写进总控 thread 历史,总控靠"读
历史"看到它 —— 这条链路脆弱(best-effort 写盘、跨 thread 持久化),一旦没落地,弱模型
(deepseek-chat)面对空历史就会**从任务语义臆造**不存在的席位(已观测到 architecture-
specialist、02714cbe… 这类假 id)。网关在每次 leader run 时本就握有真实席位 id,这里
直接把清单摆到总控本轮输入最前面并要求逐字复制,从源头根治臆造;与事后 _partition_dispatch
校验构成「源头 + 兜底」双保险。
"""
if not seats:
return ""
lines: list[str] = []
for sid, meta in seats:
name = str((meta or {}).get("name") or sid)
desc = str((meta or {}).get("description") or "").strip()
lines.append(f"- {sid}({name}):{desc}" if desc else f"- {sid}({name})")
roster = "\n".join(lines)
return (
"【本次圆桌可调度席位清单】派活时 agent_orchestration 的 agent_name 必须从下面"
"逐字符复制一个 id,严禁自行拼造、翻译角色名或凭记忆构造:\n"
f"{roster}"
)
async def _collect_delivery_status(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
seat_threads: dict[str, str],
*,
max_chars: int = 0,
) -> dict[str, tuple[bool, str, int, str]]:
"""读各席位 thread 的真实状态,客观判定「是否已交付 + 交付摘要」。
返回 ``{seat_id: (delivered, text, full_len, invalid_reason)}``。``text`` 是该席位**可见正文**
(按 ``max_chars`` 截断:派活轮取 600 字摘要、综合轮 ``max_chars<=0`` 取全文);``full_len`` 是
可见正文全文长度。判定依据:剥掉思考后是否有可见正文(不要求 md 文件);思考-only / 空串 /
断流都不算已交付。
走**轻量读**接口(``GET /last-ai-message?max_chars=N``):服务端只回最后一条 AI 消息(可截断),
不把整份 state 序列化经 loopback 传回 —— 派活轮每席位只取 600 字,大幅压低原先 ~11s 的
``payload_built``。任一席位读失败按「未交付」保守处理,绝不阻断。
这是「确定性双保险」的数据源:总控弱模型会臆造「三方已交付」,这里由网关**客观核对**每个席位
thread 的实际产出,再把结论喂回总控输入,总控以系统核对为准,不靠自己读历史猜。
"""
async def _one(sid: str, tid: str) -> tuple[str, tuple[bool, str, int, str]]:
try:
resp = await client.get(
f"{base}/api/threads/{tid}/last-ai-message",
params={"max_chars": max_chars},
headers=headers,
)
if resp.status_code != 200:
return sid, (False, "", 0, "empty")
data = resp.json()
content = str(data.get("content") or "")
full_len = int(data.get("length") or len(content))
reason = str(data.get("invalid_reason") or "")
return sid, (bool(data.get("delivered")), content, full_len, reason)
except Exception:
return sid, (False, "", 0, "empty")
results = await asyncio.gather(*(_one(sid, tid) for sid, tid in seat_threads.items()))
return dict(results)
# 交付状态摘要块里,每个席位「已交付摘要」截断到的字符数(让总控有足够信息做派活决策,
# 又远小于动辄几万字的全文)。
_STATUS_EXCERPT_CHARS = 600
def _build_seat_status_block(
seats: list[tuple[str, dict[str, Any]]],
delivery: dict[str, tuple[bool, str, int, str]] | dict[str, tuple[bool, str, int]],
*,
undelivered_count: int = 0,
) -> str:
"""席位清单 + **系统客观核对的真实交付状态**,注入总控本轮输入最前面。
在 ``_build_seat_roster_block``(只给 id+名称)基础上,叠加每个席位的 ✅已交付 / ⬜未交付
标记(来自 ``_collect_delivery_status`` 对席位 thread 的实读),并明确告诉总控「以这份为准、
不要臆造」「用户点名的 ⬜ 席位必须派活」。这是消除「臆造三方已交付直接收口」的确定性手段。
``undelivered_count`` > 0 时额外加一句硬约束:还有席位未交付,**不得**急于收口。
"""
if not seats:
return ""
lines: list[str] = []
for sid, meta in seats:
name = str((meta or {}).get("name") or sid)
item = delivery.get(sid, (False, "", 0, ""))
delivered, text, full_len = item[0], item[1], item[2]
reason = item[3] if len(item) > 3 else ""
if delivered:
# text 已是服务端按 max_chars 截好的摘要;full_len > len(text) 说明被截断,加省略号。
ellipsis = "…" if full_len > len(text) else ""
tail = f" —— 交付摘要:{text}{ellipsis}" if text else ""
lines.append(f"- {sid}({name}):✅ 已交付{tail}")
elif reason == "thinking_only":
lines.append(f"- {sid}({name}):⚠️ 无有效正文(仅思考或中断,不算交付)")
else:
lines.append(f"- {sid}({name}):⬜ 未交付")
roster = "\n".join(lines)
pending_rule = (
(
f"\n⚠️ 当前还有 {undelivered_count} 个未交付席位(⬜ 未交付或 ⚠️ 仅思考/中断)。"
"**信息尚未集齐,严禁现在收口/输出最终方案**,"
"也不得声称「各席位均已发言」。请**优先继续派活**给这些席位,把它们的真实可见正文拿到手再说综合。"
"思考过程、半截输出都不算交付。"
)
if undelivered_count > 0
else ""
)
return (
"【本次圆桌席位与真实交付状态(系统客观核对,以此为准,严禁臆造)】\n"
"派活时 agent_orchestration 的 agent_name 必须从下面逐字符复制一个 id;"
"只有标 ✅ 的才算已交付(可见正文,不要求 md 文件);标 ⬜ / ⚠️ 的都还没交付——\n"
"若用户想要的席位是 ⬜ 或 ⚠️,你**必须**派活给它拿到真实可见正文,**不得**跳过它直接收口或代答;"
"也**不得**声称任何 ⬜ / ⚠️ 席位已交付。下面只是**摘要**,做最终综合时以系统额外提供的"
"「完整交付」为准:\n"
f"{roster}"
f"{pending_rule}"
)
def _build_full_deliverables_block(
seats: list[tuple[str, dict[str, Any]]],
delivery: dict[str, tuple[bool, str, int, str]] | dict[str, tuple[bool, str, int]],
) -> str:
"""综合轮专用:把各已交付席位的**完整交付正文**拼成一块,供总控写最终方案。
只在前端标记的「综合轮」(synthesis_mode)注入,平时不进总控输入 —— 这样派活决策轮
的上下文保持精简、响应快,只有真要收口时才承担一次全文的体量。未交付席位不列入。
"""
sections: list[str] = []
for sid, meta in seats:
item = delivery.get(sid, (False, "", 0, ""))
delivered, full_text = item[0], item[1]
if not delivered or not full_text:
continue
name = str((meta or {}).get("name") or sid)
sections.append(f"━━ {name}({sid})的完整交付 ━━\n{full_text}")
if not sections:
return ""
body = "\n\n".join(sections)
return (
"【各席位完整交付正文(综合专用,做最终方案时以此为准)】\n"
"请基于下列各席位的真实完整交付,综合输出结构化的最终方案,不要遗漏关键内容,也不要臆造:\n"
f"{body}"
)
async def _leader_run(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
agent_threads: dict[str, str],
agent_name: str,
new_message: str,
skill_stop_names: list[str],
model_name: str,
seat_meta: dict[str, dict[str, Any]] | None = None,
synthesis_mode: bool = False,
suppress_pre_broadcast: bool = True,
files: list[dict[str, Any]] | None = None,
) -> AsyncIterator[str | dict[str, Any]]:
"""总控(leader)单轮:透传上游 SSE + 末尾 yield 一个 dict 状态帧(由路由层 ``json.dumps``)。
时序要点:
1. 总控输入只进入总控 thread,绝不广播到席位 thread;
2. 跑总控 thread 的 stream,yield 每一行原始 SSE;
3. 解析 messages → 澄清 / 空 dispatch(共识)/ 派活列表并返回。
``skill_stop_names`` 会与 ``agent_orchestration`` 合并,确保 middleware 在工具调用后停图。
``seat_meta``:``{seat_id: {name, description}}``。非空时,本函数会在跑总控前**实读各席位
thread 的真实交付状态**,把「席位清单 + ✅已交付/⬜未交付 + 摘要」拼到总控本轮输入最前面
(见 ``_collect_delivery_status`` / ``_build_seat_status_block``)。这是消除「总控臆造
三方已交付、一交付就收口」的确定性双保险:总控以系统客观核对为准,不靠自己读历史猜。
``synthesis_mode``:前端在「派活后让总控判断继续/综合」的轮次置真。为真时**额外**把各
已交付席位的**完整交付正文**注入本轮输入(``_build_full_deliverables_block``),供总控写
最终方案;平时(派活决策轮)只给摘要,让总控 thread 与每轮上下文保持精简、响应快。
``suppress_pre_broadcast`` 仅为兼容旧客户端保留,服务端始终按抑制处理。角色隔离不能由
客户端布尔值决定:总控 prompt 可能含 ``agent_orchestration``,进入席位历史会诱导弱模型
模仿总控加载编排能力。
"""
t0 = time.perf_counter()
# 实读各席位 thread 的真实交付状态,构造「席位清单 + 客观交付状态(摘要)」块;综合轮再额外
# 拼上各已交付席位的完整交付正文。失败/无 seat_meta 时退化为不注入(leader_input ==
# new_message),绝不阻断 leader run。这些只进总控**本轮 run 输入**,不广播给席位。
seat_status_block = ""
full_deliverables_block = ""
if seat_meta:
seat_ids = [name for name in agent_threads if name != agent_name]
if seat_ids:
# 第 1 阶段:**轻量**读各席位最后一条 AI 交付,只取 ≤_STATUS_EXCERPT_CHARS 字 ——
# 派活轮只需 ✅/⬜ + 摘要,绝不下载几万字全文,把 payload_built 从 ~11s 压下来。
delivery: dict[str, tuple[bool, str]] = {}
try:
delivery = await _collect_delivery_status(
client, base, headers, {sid: agent_threads[sid] for sid in seat_ids},
max_chars=_STATUS_EXCERPT_CHARS,
)
except Exception:
logger.warning("[leader-roster] 收集席位交付状态失败,退化为仅席位清单", exc_info=True)
undelivered = [sid for sid in seat_ids if not delivery.get(sid, (False, ""))[0]]
seats_meta_list = [(sid, seat_meta.get(sid, {})) for sid in seat_ids]
seat_status_block = _build_seat_status_block(seats_meta_list, delivery, undelivered_count=len(undelivered))
# 第 2 阶段(仅「综合轮 且 所有席位都已交付」):再**全文**取各已交付席位的交付,拼综合块。
# 还有 ⬜ 未交付席位时不注入 —— 否则"综合输出最终方案"的指令会怂恿弱模型在席位没集齐时
# 就提前收口(实测:3/5 交付就臆称"三方已发言"草草收口)。全文读只在这一轮才付出。
if synthesis_mode and not undelivered:
try:
full_delivery = await _collect_delivery_status(
client, base, headers,
{sid: agent_threads[sid] for sid in seat_ids if delivery.get(sid, (False, ""))[0]},
max_chars=0,
)
full_deliverables_block = _build_full_deliverables_block(seats_meta_list, full_delivery)
except Exception:
logger.warning("[leader-roster] 综合轮取全文交付失败,退化为仅摘要", exc_info=True)
logger.info(
"[leader-roster] %s 注入席位交付状态:%s synthesis_mode=%s undelivered=%d full_deliverables=%s",
agent_name,
{sid: delivery.get(sid, (False, ""))[0] for sid in seat_ids},
synthesis_mode,
len(undelivered),
bool(full_deliverables_block),
)
leader_input = "\n\n".join(
part for part in (seat_status_block, full_deliverables_block, new_message) if part
)
# 兼容旧请求字段但不再信任它。总控输入与派活说明始终只留在总控 thread;席位需要的上下文
# 由总控下发的具体 task、peer_deliveries 和席位交付广播提供。
_ = suppress_pre_broadcast
thread_id = agent_threads[agent_name]
# 角色参数取自单一数据源 roundtable_run_policy(leader:禁 web_search、停
# agent_orchestration 图、关 thinking + low reasoning——短决策、首字节快)。
# 调用方可再叠加自己的 skill_stop。后台进程内网关用同一份 policy,保证传参一致。
policy = roundtable_run_policy("leader")
effective_stop_names = merge_skill_stop_names(policy.skill_stop_names, skill_stop_names)
# 带了上传文件时,给总控解锁「读」类工具(leader policy 默认禁 read_file/ls/view_image),
# 否则文件清单注入了却无工具读取。写/派活类仍按 policy 封锁,不放开。
leader_excluded = policy.excluded_tools
if files:
_read_tools = {"read_file", "ls", "grep", "view_image"}
leader_excluded = [t for t in leader_excluded if t not in _read_tools]
# 模型自动容错:按 config.yaml models[] 顺序的候选链。当前模型出问题(兜底错误文案 /
# 既无正文又无派活)时切到下一个模型重跑(见循环内检测)。``_mk_body`` 用候选模型重建 body。
model_chain = fallback_model_chain(model_name) or [model_name]
model_idx = 0
current_model = model_chain[0]
def _mk_body(m: str) -> Any:
return _run_payload(
agent_name=agent_name,
model_name=m,
thread_id=thread_id,
# 注入了席位清单的输入(leader_input);seat_roster 为空时等同 new_message。
new_message=leader_input,
excluded_tools=leader_excluded,
skill_stop_names=effective_stop_names,
thinking_enabled=policy.thinking_enabled,
reasoning_effort=policy.reasoning_effort,
# 人工干预/恢复会重发 leader run 到同一总控 thread;抢占掉没结束的旧 run,避免 409。
multitask_strategy="rollback",
# 用户随干预上传的参考文件 → 总控可用 read_file 等工具读取后再派活。
files=files,
)
body = _mk_body(current_model)
t_payload_built = time.perf_counter()
logger.info(
"[leader-timing] %s payload_built=%.3fs",
agent_name,
t_payload_built - t0,
)
# 单例总控是弱模型(deepseek-chat),已观测到它把 SOUL 里的示例 id 当真实席位
# 照抄、或臆造同款 32 位 hex —— 这些 id 不在 agent_threads 里,直接漏给前端会让
# /run/stream 报 400 把整个 Step 2 打挂。这里把「本次真实可派席位」作为 roster
# (agent_threads 去掉总控自身),派活前就地校验:非法 id 丢弃 + 记日志;若**整轮
# 全是非法 id**,向总控 thread 回灌一条系统纠正消息(列出真实 id),再重跑一轮让它
# 自纠(最多重试 1 次,避免弱模型反复臆造导致死循环)。
valid_seat_ids: set[str] = {name for name in agent_threads if name != agent_name}
_MAX_DISPATCH_RETRIES = 1
attempt = 0
while True:
captured: list[str] = []
first_line_ts: float | None = None
line_count = 0
async for line, done in _stream_upstream(client, base, headers, thread_id, body):
if line is not None:
if first_line_ts is None:
first_line_ts = time.perf_counter()
logger.info(
"[leader-timing] %s first_upstream_line=%.3fs since_entry=%.3fs",
agent_name,
first_line_ts - t_payload_built,
first_line_ts - t0,
)
line_count += 1
yield line
elif done is not None:
captured = done
t_stream_done = time.perf_counter()
logger.info(
"[leader-timing] %s upstream_stream_complete=%.3fs lines=%d since_entry=%.3fs",
agent_name,
t_stream_done - (first_line_ts or t_payload_built),
line_count,
t_stream_done - t0,
)
messages = _parse_last_messages(captured)
if messages is None:
# 流被截断(多半被 multitask_strategy=rollback 抢占 / 客户端断开 / abort)。
# 这不是真正的上游错误——静默收尾,不向前端抛 502。该 run 对应的 SSE 在抢占
# 场景下通常已被前端 abortAllInFlight 放弃;即便没被放弃,前端拿到"无终态帧"
# 也只会当作空轮次,不会冒出错误条。
# WARNING 级:正常应是被 rollback 抢占;但若上游 run 因别的原因(如 DB 连接断、
# LLM 报错)截断流也会走这里 → 静默收尾会把它表现成前端"(无文本输出)"。留个
# 显眼日志,便于把"真异常被静默吞掉"和"正常抢占"区分开排查。
logger.warning(
"[leader-timing] %s upstream truncated (no values frame) — ending quietly "
"(expected on rollback preemption; otherwise check the upstream run's server-side error)",
agent_name,
)
return
dispatch_msg = _find_dispatch_message(messages)
leader_text = _flatten_content(dispatch_msg.get("content", ""))
tool_calls = dispatch_msg.get("tool_calls") or []
# 模型自动容错:本轮总控产出是「中间件兜底错误文案」或「既无正文又无派活」(模型没干活)
# → 若还有备选模型,切到下一个重跑(不消耗派活纠正预算)。先发 model_switch 帧让前端把
# 这条失败气泡重置,再用新模型重新流式。最后一个模型仍失败则照常往下走(错误文案当共识展示)。
fail_reason = is_llm_error_message(dispatch_msg)
if fail_reason is None and not (leader_text or "").strip() and not tool_calls:
fail_reason = "empty"
if fail_reason and model_idx + 1 < len(model_chain):
model_idx += 1
next_model = model_chain[model_idx]
logger.warning("[leader-fallback] %s 模型 %s 失败(%s),切换到 %s", agent_name, current_model, fail_reason, next_model)
yield {
"status": "model_switch",
"role": "leader",
"agent_name": agent_name,
"failed_model": current_model,
"next_model": next_model,
"reason": reason_text(fail_reason),
}
current_model = next_model
body = _mk_body(current_model)
continue
# 澄清分支:与 intent.py 相同字段;status 为字符串 "clarification"(非数组),
# 前端据此暂停 runOrchestration 循环,展示 Step 2 澄清条。
clarification_call = next(
(tc for tc in tool_calls if (tc.get("name") or "").lower() == "ask_clarification"),
None,
)
if clarification_call is not None:
cargs = clarification_call.get("args") or {}
question = str(cargs.get("question") or "").strip()
follow_up = _find_trailing_ai_text(messages, exclude_text=leader_text)
yield {
"status": "clarification",
"content": follow_up or question or leader_text,
"question": question,
"clarification_type": str(cargs.get("clarification_type") or ""),
"clarification_context": str(cargs.get("context") or ""),
"options": cargs.get("options") or [],
"allow_custom": bool(cargs.get("allow_custom", True)),
"allow_multiple": resolve_allow_multiple(cargs),
"agent_name": agent_name,
"model_used": current_model,
}
return
if not tool_calls:
# 空 status 数组 = 总控不再派活,前端视为共识(hasConsensus)。
yield {"status": [], "content": leader_text, "agent_name": agent_name, "model_used": current_model}
return
# 派活前按真实 roster 校验:合法保留、非法丢弃(只丢非法,不殃及同轮合法席位)。
valid_dispatch, invalid_names = _partition_dispatch(tool_calls, valid_seat_ids)
if invalid_names:
logger.warning(
"[leader-dispatch] %s 派活到 %d 个不存在的席位 id %s;真实席位=%s(attempt=%d)",
agent_name,
len(invalid_names),
invalid_names,
sorted(valid_seat_ids),
attempt,
)
# 整轮全是非法 id 且还有重试预算 → 回灌纠正消息 + 重跑一轮让总控自纠。
if invalid_names and not valid_dispatch and attempt < _MAX_DISPATCH_RETRIES:
try:
await _append_thread_message(
client,
base,
headers,
thread_id,
_build_dispatch_correction(invalid_names, valid_seat_ids),
)
except Exception as exc:
logger.warning("[leader-dispatch] 回灌纠正消息失败: %s", exc)
attempt += 1
continue
# 非法 id 已被丢弃;若 roster 校验后无合法派活(且重试已用尽),退化为空共识,
# 而不是把假 id 漏给前端再报 400。
dispatched: list[list[str]] = [[sub_name, task_text] for sub_name, task_text in valid_dispatch]
# 关键路径优先:yield 派活帧,让前端立刻启动席位 special run。总控派活说明不广播给
# 任何席位,避免席位从历史中模仿总控角色;席位间承上依靠完成交付广播 / peer_deliveries。
yield {"status": dispatched, "content": leader_text, "agent_name": agent_name, "model_used": current_model}
return
async def _special_run(
client: httpx.AsyncClient,
base: str,
headers: dict[str, str],
agent_threads: dict[str, str],
agent_name: str,
new_message: str,
skill_stop_names: list[str],
model_name: str,
*,
display_name: str = "",
no_file: bool = False,
peer_deliveries: list[dict[str, Any]] | None = None,
files: list[dict[str, Any]] | None = None,
thinking_enabled: bool | None = None,
reasoning_effort: str | None = None,
subagent_enabled: bool = False,
seat_skill_directive: bool = False,
business_code: str = "",
) -> AsyncIterator[str | dict[str, Any]]:
"""子智能体(special)单轮:执行总控下发的 task 文本,交付 content。
- ``excluded_tools`` 禁用派活/澄清等,避免席位抢总控活;
- 终态 ``status`` 为描述性字符串(非数组),``content`` 为交付正文;
- **后广播**在 yield 终态帧之后 await,保证其它席位 state 里有完成摘要,
同时尽量让前端先收到结果关闭 loading。
``no_file``:DAG 并行批次里的席位置真 —— 用 ``seat_parallel`` policy 额外禁
write_file / bash / str_replace,避免并行席位同时写沙箱打架(§4)。
``peer_deliveries``:前序席位的**完整交付** ``[{name, content}, ...]``。注入到
run context 的 ``roundtable_peer_deliveries``,供席位用 ``read_peer_delivery`` 工具
**按需**取全文(平时只看广播摘要,§12.9 传摘要按需传全文)。不进 prompt,不耗 token。
``thinking_enabled`` / ``reasoning_effort`` / ``subagent_enabled``:圆桌第二步「席位
执行模式」(快速/思考/专业问答/多智能体问答)派生的覆盖值。**仅 special 席位**接收:
前两者为 None 时回退到 policy 默认(向后兼容,旧前端不传 → 维持 thinking + medium);
``subagent_enabled`` 仅 ultra 档为 True,给席位解锁 ``task`` 子代理工具。**总控
(leader)不受影响**,始终用其快速派活策略。
"""
t0 = time.perf_counter()
thread_id = agent_threads[agent_name]
# 角色参数取自单一数据源 roundtable_run_policy(seat / seat_parallel / report)。
# report 在 seat 基础上额外禁 bash / str_replace;seat_parallel 额外禁
# write_file / bash / str_replace(理由见 roundtable_run_policy 模块注释)。
# 后台进程内网关用同一份 policy,保证「后台挂起」与「前端手动调用」传参一致。
if agent_name == REPORT_AGENT_ID:
role = "report"
elif agent_name == SUMMARY_AGENT_ID:
role = "summary"
elif agent_name == ACTION_PLAN_AGENT_ID:
role = "action_plan"
elif agent_name == DASHBOARD_AGENT_ID:
# 大屏多页数据智能体:纯「研讨成果 → report-json」提炼,禁用一切工具。
role = "dashboard"
elif agent_name == STRUCTURE_AGENT_ID:
# 结构化抽取智能体在「入库」轮:真正调用入库技能(read_file 读 SKILL.md +
# bash/write_file 落地),故走放开技能执行链路的 ``ingest`` policy。
role = "ingest"
elif no_file:
role = "seat_parallel"
else:
role = "seat"
policy = roundtable_run_policy(role)
# 带了上传文件时,给席位解锁「读」类工具(view_image 默认被禁;read_file 本就可用),
# 让席位能读取/识别用户上传的文件。写/派活类仍按 policy 封锁。
seat_excluded = policy.excluded_tools
if files:
_read_tools = {"read_file", "ls", "grep", "view_image"}
seat_excluded = [t for t in seat_excluded if t not in _read_tools]
# 弱模型补偿:仅对真正的研讨席位(seat / seat_parallel;report/summary/dashboard/ingest
# 等 Step3 单例不在此列)把其配置技能(名称 + SKILL.md 路径 + 必须先 read_file 再按其
# 流程取信息的硬指令)显式拼进本轮任务,避免弱模型无视系统提示里的技能块(见
# roundtable_seat_skills 模块说明)。默认关闭:是否注入 = 该席位 agent 的开关
# (config.yaml)OR 本次会商所选业务链条的开关(seat_skill_directive 透传)。无配置
# 技能 / 两开关都关时原样返回。
seat_message = new_message
if role in ("seat", "seat_parallel"):
seat_message = append_seat_skill_directive(agent_name, new_message, chain_enabled=seat_skill_directive)
# 角色边界放在所有技能强化之后,确保本轮最后一条指令明确:席位永远不能加载/调用编排能力。
seat_message = append_seat_role_boundary(seat_message)
# 深链接业务规范(仅 summary 单例):把该业务的「业务链完整性核对」指令拼进本轮任务,强制总结报告
# 核对产出相对业务链完不完整。business_code 为空(普通会商)/ 非写实业务 → directive 为 "",原样不变。
elif role == "summary" and business_code:
directive = build_summary_directive(business_code)
if directive:
seat_message = f"{new_message}\n\n---\n\n{directive}"
# 模型自动容错:按 config.yaml models[] 顺序的候选链。某模型出问题(兜底错误文案;
# 真正的研讨席位还含「交付为空」)→ 发 model_switch 帧让前端重置该席位气泡,再用下一个
# 模型重跑。Step3 单例(report/summary/dashboard/ingest)的产物是落盘文件,正文可能很短,
# 故**不**把「正文为空」当失败(与后台 run_report/summary 的 validate=None 一致),只在真正
# 的模型错误时换模型。
model_chain = fallback_model_chain(model_name) or [model_name]
empty_is_failure = role in ("seat", "seat_parallel")
def _mk_body(m: str) -> Any:
return _run_payload(
agent_name=agent_name,
model_name=m,
thread_id=thread_id,
new_message=seat_message,
excluded_tools=seat_excluded,
skill_stop_names=merge_skill_stop_names(policy.skill_stop_names, skill_stop_names),
# 席位执行模式覆盖 policy 的 thinking/reasoning(None → 回退 policy 默认);
# subagent 仅 ultra 档置真。工具裁剪(excluded_tools)仍完全由 policy 决定。
thinking_enabled=(policy.thinking_enabled if thinking_enabled is None else thinking_enabled),
reasoning_effort=(policy.reasoning_effort if reasoning_effort is None else reasoning_effort),
subagent_enabled=subagent_enabled,
# ultra 席位才会派子智能体 → 才需要 custom 流回传子任务实时进度(task_running 等)。
stream_custom=subagent_enabled,
# 干预/重新派活会把同一席位 thread 再跑一次;抢占掉没结束的旧 run,避免 409。
multitask_strategy="rollback",
# 前序席位完整交付 → run context,供 read_peer_delivery 工具按需取全文(§12.9)。
extra_context=(
{"roundtable_peer_deliveries": peer_deliveries} if peer_deliveries else None
),
# 用户随干预上传的参考文件 → 席位可用 read_file 等工具读取。
files=files,
)
t_payload_built = time.perf_counter()
logger.info(
"[special-timing] %s payload_built=%.3fs",
agent_name,
t_payload_built - t0,
)
# 优先用 display_name(中文友好,例如"情报收集")提示 leader,id 放括号备查;
# display_name 缺失时退化为仅 agent_name —— 与旧行为一致。
speaker_label = f"{display_name}({agent_name})" if display_name else agent_name
for idx, candidate in enumerate(model_chain):
body = _mk_body(candidate)
captured: list[str] = []
first_line_ts: float | None = None
line_count = 0
async for line, done in _stream_upstream(client, base, headers, thread_id, body):
if line is not None:
if first_line_ts is None:
first_line_ts = time.perf_counter()
logger.info(
"[special-timing] %s first_upstream_line=%.3fs since_entry=%.3fs",
agent_name,
first_line_ts - t_payload_built,
first_line_ts - t0,
)
line_count += 1
yield line
elif done is not None:
captured = done
t_stream_done = time.perf_counter()
logger.info(
"[special-timing] %s upstream_stream_complete=%.3fs lines=%d since_entry=%.3fs",
agent_name,
t_stream_done - (first_line_ts or t_payload_built),
line_count,
t_stream_done - t0,
)
messages = _parse_last_messages(captured)
if messages is None:
# 流被截断(网断 / rollback 抢占)。不抛 502,但席位路径要明确告诉前端:本轮不算交付,
# 避免总控把 checkpoint 里残留的思考/半截当成已完成。
logger.warning(
"[special-timing] %s upstream truncated (no values frame) — ending as invalid_delivery "
"(expected on rollback preemption; otherwise check the upstream run's server-side error)",
agent_name,
)
if empty_is_failure:
yield {
"status": "invalid_delivery",
"reason": "truncated",
"content": "",
"agent_name": agent_name,
"model_used": candidate,
}
return
# Sub-agent's final deliverable is visible body after the last human (this turn).
last_msg: dict[str, Any] = {}
for msg in reversed(messages):
if isinstance(msg, dict) and (msg.get("type") or "").lower() in ("ai", "aimessage", "aimessagechunk"):
last_msg = msg
break
content = _flatten_content(last_msg.get("content", ""))
visible = visible_seat_text(content)
if not visible:
visible = _visible_delivery_from_messages(messages)
delivery_kind = classify_seat_delivery(content) if content.strip() else ("valid" if visible else "empty")
# 模型失败检测:兜底错误文案;研讨席位再加「无可见正文」(思考不算)。还有备选模型则换。
fail_reason = is_llm_error_message(last_msg)
if fail_reason is None and empty_is_failure and delivery_kind != "valid":
fail_reason = "thinking_only" if delivery_kind == "thinking_only" else "empty"
if fail_reason and idx + 1 < len(model_chain):
next_model = model_chain[idx + 1]
logger.warning("[special-fallback] %s 模型 %s 失败(%s),切换到 %s", agent_name, candidate, fail_reason, next_model)
yield {
"status": "model_switch",
"role": role,
"agent_name": agent_name,
"failed_model": candidate,
"next_model": next_model,
"reason": reason_text(fail_reason),
}
continue
if empty_is_failure and delivery_kind != "valid":
logger.warning("[special-fallback] %s 所有模型均未产出可见正文(%s)", agent_name, delivery_kind)
yield {
"status": "invalid_delivery",
"reason": delivery_kind,
"content": "",
"agent_name": agent_name,
"model_used": candidate,
}
return
# 成功:发交付终态帧 + detached 后广播(只广播可见正文,不含思考)。
_prefix = f"子智能体{speaker_label}完成{new_message}工作,交付内容为:"
brief_msg = _truncate_broadcast_content(_prefix, visible or content)
yield {
"status": f"子智能体{speaker_label}完成{new_message}工作",
"content": visible or content,
"agent_name": agent_name,
"model_used": candidate,
}
_spawn_detached_broadcast(base, headers, brief_msg, agent_threads, [agent_name])
return
# ---------------------------------------------------------------------------
# POST /api/multi-agent/run/stream — leader / special 单轮 SSE
# ---------------------------------------------------------------------------
@router.post("/run/stream")
async def run_multi_agent_stream(request: Request, payload: dict = Body(...)) -> StreamingResponse:
"""驱动总控或子智能体的一轮 LangGraph run。
请求体::
{
"agent_type": "leader" | "special",
"d_agent_thread_id": { "<agent_id>": "<thread_id>", ... }, // init_done.thread_ids
"agent_name": "<本次运行的 agent_id>",
"new_message": "<用户消息或派活 task 文本>",
"model": "<可选>", // 不传由 lead_agent 回退到 config.yaml models[0]
"skill_stop_names": [] // 可选,leader 一般留空,由网关注入 agent_orchestration
}
响应:上游 SSE 行透传 + 末尾一条 ``data:`` JSON(leader 为 dict,special 同上)。
客户端应使用 AbortSignal 支持「挂起协商」时取消 in-flight 请求。
"""
agent_type: str = str(payload.get("agent_type") or "leader").lower()
agent_threads: dict[str, str] = payload.get("d_agent_thread_id") or {}
agent_name: str = str(payload.get("agent_name") or "").lower()
# 人类可读名,仅 special 用 —— 后端把它拼进 post-broadcast 文本,这样 leader
# 下一轮看到的是"子智能体[中文名](id)完成..."而非"子智能体[UUID]完成...";
# LLM 对 UUID 不友好,只用 id 会导致总控认不出已完成的子智能体,无限循环派活。
# 缺失时 fallback 到 agent_name,旧前端不传也不破坏。
display_name: str = str(payload.get("display_name") or "").strip()
new_message: str = str(payload.get("new_message") or "")
skill_stop_names: list[str] = payload.get("skill_stop_names") or []
# 用户随干预消息上传的参考文件(已经过 /api/threads/{thread_id}/uploads)。
# 前端传 [{filename, size, path, status}, ...],透传给 _run_payload →
# additional_kwargs.files → UploadsMiddleware 注入清单 + 解锁 read_file 等工具。
_files_raw = payload.get("files")
files: list[dict[str, Any]] = _files_raw if isinstance(_files_raw, list) else []
# 综合轮标记(仅 leader 用):前端在「派活后让总控判断继续/综合」的轮次置真,网关据此
# 额外把各已交付席位的完整交付正文注入总控输入,供其写最终方案。缺省 False(派活决策轮
# 只给摘要,保持上下文精简)。旧前端不传 → False,行为安全退化。
synthesis_mode: bool = bool(payload.get("synthesis_mode") or False)
# 并行禁写文件标记(仅 special 用):前端 runDagOrchestration 在真并行 stage(≥2 席位)
# 置真,网关据此用 seat_parallel policy 额外禁 write_file/bash/str_replace(§4)。
# 兼容 no_file 别名;缺省 False → 普通 seat,旧前端不传也不破坏。
no_file: bool = bool(payload.get("parallel_no_file") or payload.get("no_file") or False)
# 旧客户端兼容字段。服务端已强制隔离总控 thread,因此无论请求值为何都不会执行 leader→seat
# 广播;仍解析并透传只是保持调用签名稳定。
suppress_pre_broadcast: bool = bool(payload.get("suppress_pre_broadcast", True))
# 前序席位完整交付(仅 special 用):前端 DAG stage>0 席位带上此前各阶段的全文,网关注入
# run context,供 read_peer_delivery 工具按需取全文(平时只看广播摘要,§12.9)。规整成
# [{name, content}] 且过滤空项;缺省 None。
peer_deliveries: list[dict[str, Any]] | None = None
_pd_raw = payload.get("peer_deliveries")
if isinstance(_pd_raw, list):
peer_deliveries = [
{"name": str(p.get("name") or ""), "content": str(p.get("content") or "")}
for p in _pd_raw
if isinstance(p, dict) and str(p.get("name") or "").strip() and str(p.get("content") or "").strip()
] or None
# 不写死任何默认模型;空串会让 lead_agent._resolve_model_name() 回退到
# config.yaml 的 models[0],与 /api/ai-writing 等其它路由一致。
model_name: str = str(payload.get("model") or "")
# 圆桌第二步「席位执行模式」(快速/思考/专业问答/多智能体问答)派生参数。**仅 special
# 席位消费**:覆盖 seat policy 的 thinking/reasoning。thinking_enabled/reasoning_effort
# 缺省(None/空)→ 后端按 policy 默认执行(向后兼容,旧前端不传维持 thinking + medium);
# subagent_enabled 仅 ultra 档为真,给席位解锁 task 子代理工具。leader 不消费这些字段。
_te_raw = payload.get("thinking_enabled")
seat_thinking_enabled: bool | None = bool(_te_raw) if _te_raw is not None else None
_re_raw = payload.get("reasoning_effort")
seat_reasoning_effort: str | None = _re_raw if isinstance(_re_raw, str) and _re_raw else None
seat_subagent_enabled: bool = bool(payload.get("subagent_enabled") or False)
# 业务链条级「技能识别强化」开关(仅 special 席位消费):前端读所选业务链条的同名配置
# 后逐席位透传。与该席位 agent 自身的 config.yaml 开关取 OR;缺省 False(默认关)。
seat_skill_directive: bool = bool(payload.get("seat_skill_directive") or False)
# 深链接业务码(rwfx→6BF 等)。仅 summary 单例消费:非空 → 按业务链抽取规范强制注入「完整性核对」
# 指令,让总结报告核对产出相对业务链完不完整。普通会商为空 → 不注入。
business_code: str = str(payload.get("business_code") or "")
if not agent_name or agent_name not in agent_threads:
raise HTTPException(
status_code=400,
detail=f"agent_name '{agent_name}' is missing from d_agent_thread_id",
)
_validate_agent_id(agent_name, "agent_name")
for key in agent_threads:
_validate_agent_id(key, "d_agent_thread_id key")
base = _loopback_base(request)
headers = _auth_headers(request)
# 总控派活前,取本次各席位的元数据(id+名称+描述);_leader_run 会据此**实读各席位
# thread 的真实交付状态**,把「席位清单 + ✅/⬜交付状态」拼进总控本轮输入最前面 ——
# 既让弱模型每轮都能逐字复制真实 id,又以系统客观核对的交付状态压制臆造。元数据走同进程
# 批量查;任一席位查不到/不可见就退化为仅 id 清单,绝不因此阻断 leader run。
seat_meta: dict[str, dict[str, Any]] | None = None
if agent_type == "leader":
seat_ids = [k for k in agent_threads if k != agent_name]
if seat_ids:
try:
seat_meta = await _get_agents_batch_direct(request, seat_ids)
except Exception:
logger.warning("[leader-roster] 取席位元数据失败,退化为仅 id 清单", exc_info=True)
seat_meta = {sid: {"id": sid, "name": sid, "description": ""} for sid in seat_ids}
async def stream_generator() -> AsyncIterator[str]:
# TCP loopback 到 127.0.0.1:<gateway port>。client 配置见 _make_loopback_client:
# trust_env=False 防 HTTP_PROXY 拦截,timeout=None 容纳 SSE 长连接。
async with _make_loopback_client(request) as client:
try:
if agent_type == "leader":
gen = _leader_run(
client, base, headers, agent_threads, agent_name, new_message, skill_stop_names, model_name,
seat_meta=seat_meta,
synthesis_mode=synthesis_mode,
suppress_pre_broadcast=suppress_pre_broadcast,
files=files,
)
else:
gen = _special_run(
client, base, headers, agent_threads, agent_name, new_message, skill_stop_names, model_name,
display_name=display_name,
no_file=no_file,
peer_deliveries=peer_deliveries,
files=files,
thinking_enabled=seat_thinking_enabled,
reasoning_effort=seat_reasoning_effort,
subagent_enabled=seat_subagent_enabled,
seat_skill_directive=seat_skill_directive,
business_code=business_code,
)
async for chunk in gen:
if isinstance(chunk, dict):
yield f"data: {json.dumps(chunk, ensure_ascii=False)}\n\n"
else:
yield chunk if chunk.endswith("\n") else chunk + "\n"
except HTTPException as exc:
logger.warning("multi-agent stream HTTPException: %s %s", exc.status_code, exc.detail)
await record_foreground_diag(
request, stage=_stage_for_agent(agent_type, agent_name), level="error",
event="run_http_error",
message=f"「{agent_name}」运行失败 [{exc.status_code}]:{exc.detail}",
detail={"status_code": exc.status_code, "detail": str(exc.detail), "agent_type": agent_type},
agent_id=agent_name, agent_name=display_name or agent_name,
)
err_frame = {
"status": "error",
"content": f"[{exc.status_code}] {exc.detail}",
"agent_name": agent_name,
}
yield f"data: {json.dumps(err_frame, ensure_ascii=False)}\n\n"
except Exception as exc:
logger.exception("multi-agent stream unexpected failure")
await record_foreground_diag(
request, stage=_stage_for_agent(agent_type, agent_name), level="error",
event="run_unexpected",
message=f"「{agent_name}」运行未预期异常:{exc!r}",
detail={"type": type(exc).__name__, "error": str(exc), "agent_type": agent_type},
agent_id=agent_name, agent_name=display_name or agent_name,
)
err_frame = {
"status": "error",
"content": f"unexpected: {exc!r}",
"agent_name": agent_name,
}
yield f"data: {json.dumps(err_frame, ensure_ascii=False)}\n\n"
return StreamingResponse(
stream_generator(),
media_type="text/event-stream",
headers={
"Cache-Control": "no-cache",
"X-Accel-Buffering": "no",
"Connection": "keep-alive",
},
)