2224 lines
111 KiB
Python
2224 lines
111 KiB
Python
"""圆桌规划 · Step 2「多智能体研讨」Gateway 路由。
|
||
|
||
前端页面:``RoundtablePlanningPage`` Step 2
|
||
前端模块:``frontend-web/src/roundtable-planning/api/multi-agent.ts``
|
||
|
||
本模块把原 ``newpython/OUT.py`` 演示编排进 FastAPI Gateway,通过 **HTTP loopback**
|
||
(``127.0.0.1`` + 调用方 ``Authorization``)复用既有 ``/api/threads``、
|
||
``/api/agents``、``/api/threads/{id}/runs/stream``,不再硬编码 token。
|
||
|
||
对外接口
|
||
--------
|
||
1. ``POST /api/multi-agent/init``(SSE)
|
||
- 为每个研讨席位(``agent_names``)并行创建 thread,按完成顺序推送
|
||
``seat_ready``;
|
||
- 动态创建**当次会话**的总控协调智能体(``main_agent_name``,如
|
||
``roundtable-coordinator-xxx``),写入席位列表 SOUL + ``agent_orchestration`` skill;
|
||
- 结束帧 ``init_done`` 携带 ``thread_ids: { agent_name: thread_id, ... }``,
|
||
含协调智能体自身 thread。
|
||
|
||
2. ``POST /api/multi-agent/run/stream``(SSE)
|
||
- ``agent_type=leader``:总控一轮——可能派活、可能澄清、可能共识(空 dispatch);
|
||
- ``agent_type=special``:单席位子智能体执行一轮交付。
|
||
|
||
跨席位「广播」语义
|
||
------------------
|
||
通过 ``_append_thread_message`` 向其它 thread 的 checkpoint ``messages`` 追加
|
||
**只读上下文**的 human 消息(非触发 run),使各席位 LangGraph 状态里能看到
|
||
子 agent 完成摘要。总控输入、派活 prompt 和派活说明只留在总控 thread,
|
||
绝不广播给席位;前端编排循环仍靠显式多次 ``/run/stream`` 驱动各 agent 真正推理。
|
||
|
||
Leader 终态帧(前端 ``isFinalStatusFrame``)
|
||
-------------------------------------------
|
||
- ``status: [[agent_name, task], ...]`` — 派活列表,前端并行 special run;
|
||
- ``status: []`` — 无 tool_calls,视为共识,可进 Step 3;
|
||
- ``status: "clarification"`` — 总控 ``ask_clarification``,前端暂停编排展示澄清条;
|
||
- ``status: "子智能体 X 完成..."`` — special 完成(字符串状态);
|
||
- ``status: "invalid_delivery"`` — 席位本轮无可见正文(仅思考 / 断流 / 空串),不算交付;
|
||
- ``status: "error"`` — 失败。
|
||
|
||
共享导出
|
||
--------
|
||
``intent.py`` / ``recommend.py`` 从此模块 import 底层流式与消息解析函数,
|
||
勿在 intent/recommend 中重复实现 loopback 逻辑。
|
||
|
||
相关文档:``frontend-web/docs/multi-agent-backend-dev.md``
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import asyncio
|
||
import json
|
||
import logging
|
||
import os
|
||
import re
|
||
import time
|
||
import traceback
|
||
import uuid
|
||
from collections.abc import AsyncIterator
|
||
from typing import Any
|
||
|
||
import httpx
|
||
from fastapi import APIRouter, Body, HTTPException, Request
|
||
from fastapi.responses import StreamingResponse
|
||
from langgraph.checkpoint.base import empty_checkpoint
|
||
|
||
from app.gateway.roundtable_diag import record_foreground_diag
|
||
from app.gateway.roundtable_model_fallback import fallback_model_chain, is_llm_error_message, reason_text
|
||
from app.gateway.roundtable_run_policy import merge_skill_stop_names, roundtable_run_policy
|
||
from app.gateway.roundtable_seat_skills import append_seat_role_boundary, append_seat_skill_directive
|
||
from app.gateway.routers._roundtable_seed import (
|
||
ACTION_PLAN_AGENT_ID,
|
||
COORDINATOR_AGENT_ID,
|
||
DASHBOARD_AGENT_ID,
|
||
REPORT_AGENT_ID,
|
||
STRUCTURE_AGENT_ID,
|
||
SUMMARY_AGENT_ID,
|
||
ensure_roundtable_functional_agents,
|
||
)
|
||
from app.gateway.services import RUN_SCOPE_ORCHESTRATION_CHILD, internal_run_scope_headers
|
||
from deerflow.agents.roundtable_orchestrator.delivery import classify_seat_delivery, visible_seat_text
|
||
from deerflow.agents.roundtable_orchestrator.extraction_spec import build_summary_directive
|
||
from deerflow.config import get_app_config
|
||
from deerflow.runtime.user_context import get_effective_user_id
|
||
from deerflow.tools.builtins.clarification_utils import resolve_allow_multiple
|
||
from deerflow.utils.time import now_iso
|
||
|
||
logger = logging.getLogger(__name__)
|
||
router = APIRouter(prefix="/api/multi-agent", tags=["multi-agent"])
|
||
|
||
|
||
def _stage_for_agent(agent_type: str, agent_name: str) -> str:
|
||
"""把一次 /run/stream 的 (agent_type, agent_name) 映射成诊断日志的 stage。
|
||
|
||
Step3 的四个内置单例(report/summary/dashboard/structure)也走 special 路径,
|
||
据 agent_name 归到对应 step3_* 阶段;其余 leader→step2_leader、special→step2_seat。
|
||
"""
|
||
if agent_name == REPORT_AGENT_ID:
|
||
return "step3_report"
|
||
if agent_name == SUMMARY_AGENT_ID:
|
||
return "step3_summary"
|
||
if agent_name == ACTION_PLAN_AGENT_ID:
|
||
return "position_action_plan"
|
||
if agent_name == DASHBOARD_AGENT_ID:
|
||
return "step3_dashboard"
|
||
if agent_name == STRUCTURE_AGENT_ID:
|
||
return "step3_structure"
|
||
return "step2_leader" if agent_type == "leader" else "step2_seat"
|
||
|
||
|
||
# 圆桌(intent / recommend / 席位 / 协调)创建的 thread 都是功能型内部线程,
|
||
# 不是用户的普通对话。打上这个 metadata,前端「最近的对话」列表(recent-chat-list.tsx)
|
||
# 据 `metadata.thread_type` 过滤掉,与 scheduler / bootstrap 系统会话同一套机制。
|
||
_ROUNDTABLE_THREAD_METADATA: dict[str, Any] = {"thread_type": "roundtable", "system": True}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 共享:异常诊断器(intent/recommend/multi-agent 共用)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def _diagnose_exception(exc: BaseException, *, context: str = "") -> dict[str, Any]:
|
||
"""把后端异常转成结构化诊断 dict,直接给前端展示用。
|
||
|
||
设计目标:让运维 / 前端开发不用翻 backend 日志就能看到根因和处置建议。
|
||
返回字段:
|
||
- code: 短稳定分类(loopback_connect_failed / upstream_connect_failed / ...)
|
||
- message: 一行人类可读
|
||
- exception_type: 表层异常类名
|
||
- exception_repr: repr(exc)
|
||
- root_type / root_repr: 沿 __cause__ 链找到的最深异常(若与表层不同)
|
||
- hint: 针对常见失败模式的可执行建议(中文)
|
||
- trace_tail: traceback 最后 ~12 行(避免响应体过大)
|
||
- context: 调用方标签(如 "intent_init")
|
||
"""
|
||
exc_type = type(exc).__name__
|
||
exc_repr = repr(exc)
|
||
|
||
# 沿 __cause__ 链找最深异常:httpx 经常把 ConnectError 包成 RemoteProtocolError 等。
|
||
root: BaseException = exc
|
||
seen: set[int] = {id(root)}
|
||
while getattr(root, "__cause__", None) is not None and id(root.__cause__) not in seen:
|
||
root = root.__cause__ # type: ignore[assignment]
|
||
seen.add(id(root))
|
||
|
||
code = "unexpected"
|
||
hint = "查看 trace_tail 最后一帧定位代码位置;若信息不足请到 backend 日志找完整 traceback。"
|
||
|
||
if isinstance(root, httpx.ConnectError):
|
||
target = str(root).strip().lower()
|
||
# loopback 走真 TCP 到 127.0.0.1:<gateway port>;失败 = Gateway 进程没起 / 端口被占。
|
||
# outbound 失败(其它 host)= LLM 提供商不可达。
|
||
if "127.0.0.1" in target or "localhost" in target or "loopback" in target:
|
||
code = "loopback_connect_failed"
|
||
hint = (
|
||
"本地 Gateway loopback 连接失败:确认后端进程正在监听配置的端口。"
|
||
"端口配置优先级(高→低):env DEER_FLOW_GATEWAY_PORT > config.yaml 的 "
|
||
"gateway.port > 默认 8001。若部署机改了端口,在 config.yaml 加 "
|
||
"`gateway:\\n port: <端口>` 即可。"
|
||
)
|
||
else:
|
||
code = "upstream_connect_failed"
|
||
hint = (
|
||
"外网/上游服务连接失败:大概率是 LLM 提供商不可达"
|
||
"(如 api.deepseek.com 在内网被防火墙拦截)。"
|
||
"把 config.yaml 里所有 base_url 改成内网可达的 LLM endpoint,"
|
||
"或加白名单/正向代理(HTTPS_PROXY)。"
|
||
)
|
||
elif isinstance(root, httpx.TimeoutException):
|
||
code = "upstream_timeout"
|
||
hint = (
|
||
"上游请求超时:网络抖动 / 模型推理过慢。"
|
||
"先 curl <base_url>/v1/models 确认可达,再看模型是否过载或 max_tokens 过大。"
|
||
)
|
||
elif isinstance(root, httpx.HTTPStatusError):
|
||
code = "upstream_http_error"
|
||
status = getattr(getattr(root, "response", None), "status_code", "?")
|
||
hint = (
|
||
f"上游接口返回 HTTP {status}:可能是 API key 失效、模型名不对、"
|
||
"或目标服务自身报错。查看 backend 日志里 response.text 拿详细错误体。"
|
||
)
|
||
elif isinstance(root, FileNotFoundError):
|
||
code = "file_not_found"
|
||
hint = (
|
||
f"找不到路径:{root}。常见原因:内置 agent 目录(.deer-flow/agents/<id>/)缺失,"
|
||
"或 SOUL.md 没拷过来;详见后端开发文档 §11.7。"
|
||
)
|
||
elif isinstance(root, KeyError):
|
||
code = "key_error"
|
||
hint = (
|
||
f"缺少 key:{root}。若 key 是 agent_name/agent_id 类:大概率内置 agent 目录缺失;"
|
||
"若是配置 key:检查 config.yaml schema。"
|
||
)
|
||
elif isinstance(root, ValueError) and "model" in str(root).lower():
|
||
code = "no_models_configured"
|
||
hint = "config.yaml 没有任何 models[],或所有条目都解析失败。检查 yaml 语法与 $ENV 引用。"
|
||
|
||
trace_lines = traceback.format_exception(type(exc), exc, exc.__traceback__)
|
||
trace_tail = "".join(trace_lines).splitlines()[-12:]
|
||
|
||
return {
|
||
"code": code,
|
||
"message": f"{exc_type}: {exc}" if str(exc) else exc_type,
|
||
"exception_type": exc_type,
|
||
"exception_repr": exc_repr,
|
||
"root_type": type(root).__name__ if root is not exc else None,
|
||
"root_repr": repr(root) if root is not exc else None,
|
||
"hint": hint,
|
||
"trace_tail": trace_tail,
|
||
"context": context or None,
|
||
}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Loopback 辅助函数(Gateway → Gateway,同进程)
|
||
#
|
||
# 圆桌三路由(intent / recommend / multi_agent)需要从一个 FastAPI handler 里
|
||
# 调用本机 Gateway 的另一个 endpoint(`/api/threads`、`/api/agents`、
|
||
# `/api/threads/{id}/runs/stream`)。这里用真 TCP loopback 到本机的 127.0.0.1
|
||
# + Gateway 监听端口。
|
||
#
|
||
# ⚠️ 历史踩坑(三个) ⚠️
|
||
# 1. 直接走系统代理:开发机若设了 HTTP_PROXY / HTTPS_PROXY,httpx 默认会把
|
||
# 127.0.0.1 也走代理 → 上游代理不可达,40s 超时(典型日志 headers_in=42.160s)。
|
||
# → 修法:client 用 `trust_env=False`,强制忽略代理 env vars。
|
||
#
|
||
# 2. 用 request.url.netloc 拼 base URL:Nginx 反代部署时,客户端打到 nginx 的
|
||
# public host(如 example.com),nginx 把请求转给 gateway 时 Host header
|
||
# 通常是 `gateway:8001` 或 public host;后者 httpx 会真的去 DNS 解析,
|
||
# 在 gateway 容器内可能不可达 → 连接失败。
|
||
# → 修法:**写死 127.0.0.1**,完全无视 request.url.netloc。
|
||
#
|
||
# 3. 一度尝试用 httpx.ASGITransport(app=request.app) 进程内直调,看似一劳永逸,
|
||
# 但 httpx 0.28 的 ASGITransport 对 streaming response **整段 buffer**,
|
||
# 一直收齐所有 body chunks 才一次性把响应给消费方 → SSE 完全失效,前端表现
|
||
# 为"调度阶段很长 + 流式结果一次性出现"。已弃用。
|
||
# → 详见:git log 6e3a07b(那次改 ASGI 引入)及随后回退。
|
||
#
|
||
# 端口策略(优先级,从高到低):
|
||
# 1. 环境变量 DEER_FLOW_GATEWAY_PORT
|
||
# ── 部署侧紧急覆盖通道,适合 docker-compose / k8s 注入或临时调试
|
||
# 2. config.yaml 的 `gateway.port`
|
||
# ── 部署侧**推荐**的配置方式,改完不需要重启二进制(get_app_config 有 mtime
|
||
# 热加载,但仍建议跑一次 reload 以确保所有 worker 拿到新值)
|
||
# 3. request.url.port
|
||
# ── 浏览器直连 gateway:8001 时拿得到;走 Nginx 80/443 反代时会落到下一档
|
||
# 4. 默认 8001(与 `make dev` / `make gateway` / `scripts/start-all.sh` 写死的一致)
|
||
#
|
||
# 端口在哪里改(给运维 / 部署侧的提示):
|
||
# - yaml 配置:打开 config.yaml,加 / 改:
|
||
# gateway:
|
||
# port: 9999
|
||
# - 环境变量:
|
||
# export DEER_FLOW_GATEWAY_PORT=9999
|
||
#
|
||
# 修改这两个函数时务必跑一次 SSE 流式 smoke 验证(见 multi-agent-backend-dev.md
|
||
# §13 测试建议),确认 chunks 是逐个到达而不是一次喷出。
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
# 写死的最终 fallback。**不要**为了改端口而修改这个常量——优先用 config.yaml /
|
||
# env 这两层(它们在部署侧不需要改源码即可生效)。这个值只在 yaml 缺失 + env
|
||
# 没设 + 请求里也拿不到合法 port 时才会用到。
|
||
_DEFAULT_GATEWAY_PORT = 8001
|
||
|
||
|
||
def _resolve_gateway_port(request: Request) -> int:
|
||
"""按四档 fallback 解析 Gateway 监听端口。详见模块顶部「端口策略」。"""
|
||
# 第 1 档:环境变量。允许非法值(garbage),只警告并继续下一档,不抛异常。
|
||
env_port = os.environ.get("DEER_FLOW_GATEWAY_PORT", "").strip()
|
||
if env_port:
|
||
try:
|
||
port = int(env_port)
|
||
if 1 <= port <= 65535:
|
||
return port
|
||
raise ValueError("out of range")
|
||
except ValueError:
|
||
logger.warning(
|
||
"DEER_FLOW_GATEWAY_PORT=%r 不是合法端口,fallback 到 config.yaml / request / %d",
|
||
env_port,
|
||
_DEFAULT_GATEWAY_PORT,
|
||
)
|
||
|
||
# 第 2 档:config.yaml 的 gateway.port。GatewayConfig.port 的 pydantic 校验
|
||
# 已经保证它在 1-65535 范围内,无需再次校验。读 config 失败时降级,不阻断业务。
|
||
try:
|
||
configured = get_app_config().gateway.port
|
||
if configured:
|
||
return configured
|
||
except Exception as exc:
|
||
# 极端场景:config.yaml 不存在或解析失败。这种情况下 gateway 本身也起不来,
|
||
# loopback 端口"猜错"也已经无所谓。仅记 debug 日志方便排查。
|
||
logger.debug("read gateway.port from app config failed: %s", exc)
|
||
|
||
# 第 3 档:从当前请求拿 port。浏览器直连 gateway 时是真实监听端口;
|
||
# 走 Nginx 反代时通常是 80/443(public port),那就跳过,落到最终默认。
|
||
inbound_port = request.url.port
|
||
if inbound_port is not None and inbound_port not in (80, 443):
|
||
return inbound_port
|
||
|
||
# 第 4 档:写死默认。
|
||
return _DEFAULT_GATEWAY_PORT
|
||
|
||
|
||
def _loopback_base(request: Request) -> str:
|
||
"""返回 `http://127.0.0.1:<port>`,供 loopback 调用拼 URL。
|
||
|
||
强制 127.0.0.1 是为了避免 Nginx 反代场景下读到经过 Host 改写的 netloc。
|
||
端口按 `_resolve_gateway_port` 的四档 fallback 解析。
|
||
"""
|
||
return f"http://127.0.0.1:{_resolve_gateway_port(request)}"
|
||
|
||
|
||
def _make_loopback_client(request: Request) -> httpx.AsyncClient:
|
||
"""构造同进程 loopback 用的 httpx.AsyncClient(真 TCP)。
|
||
|
||
所有圆桌路由(intent / recommend / multi_agent)的 loopback 调用必须走这里,
|
||
保持配置一致。
|
||
- `trust_env=False`:**关键**,忽略 HTTP_PROXY / HTTPS_PROXY,防代理拦截 127.0.0.1
|
||
- `timeout=None`:SSE 长连接需要,LangGraph runs/stream 可能跑数十秒
|
||
- `request` 参数当前未被使用,保留是为了:(a) 未来若需要从 request 取部署
|
||
上下文(如多租户端口隔离)可以无侵入扩展;(b) 与调用方现有签名兼容。
|
||
"""
|
||
return httpx.AsyncClient(
|
||
timeout=None,
|
||
trust_env=False,
|
||
)
|
||
|
||
|
||
def _auth_headers(request: Request) -> dict[str, str]:
|
||
token = request.headers.get("authorization")
|
||
if not token:
|
||
raise HTTPException(status_code=401, detail="Missing Authorization header")
|
||
return {
|
||
"Authorization": token,
|
||
"Content-Type": "application/json",
|
||
**internal_run_scope_headers(),
|
||
}
|
||
|
||
|
||
async def _create_thread(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
*,
|
||
metadata: dict[str, Any] | None = None,
|
||
) -> str:
|
||
"""``POST /api/threads`` → 新 thread_id(intent/recommend/init 席位均用)。
|
||
|
||
``metadata`` 写入 thread 元数据,圆桌调用方传 ``_ROUNDTABLE_THREAD_METADATA``
|
||
把这些功能型线程标记成可被前端「最近的对话」过滤掉的系统会话。
|
||
"""
|
||
body: dict[str, Any] = {}
|
||
if metadata:
|
||
body["metadata"] = metadata
|
||
resp = await client.post(f"{base}/api/threads", json=body, headers=headers)
|
||
resp.raise_for_status()
|
||
return resp.json()["thread_id"]
|
||
|
||
|
||
async def _get_agent(client: httpx.AsyncClient, base: str, headers: dict[str, str], agent_id: str) -> dict[str, Any]:
|
||
resp = await client.get(f"{base}/api/agents/{agent_id}", headers=headers)
|
||
resp.raise_for_status()
|
||
return resp.json()
|
||
|
||
|
||
async def _create_agent(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
*,
|
||
name: str,
|
||
description: str,
|
||
model: str | None,
|
||
skills: list[str] | None,
|
||
soul: str,
|
||
agent_id: str | None = None,
|
||
) -> tuple[bool, dict[str, Any] | str]:
|
||
payload: dict[str, Any] = {"name": name, "description": description, "soul": soul}
|
||
if agent_id is not None:
|
||
payload["id"] = agent_id
|
||
if model is not None:
|
||
payload["model"] = model
|
||
if skills is not None:
|
||
payload["skills"] = skills
|
||
resp = await client.post(f"{base}/api/agents", json=payload, headers=headers)
|
||
if resp.status_code == 201:
|
||
return True, resp.json()
|
||
return False, resp.text
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 同进程直调 helpers(仅供 init_multi_agent 用,不要扩散到其它地方)
|
||
#
|
||
# init 阶段串了 N+1 个 thread create + N 个 GET /api/agents + 1 个 POST /api/agents,
|
||
# 用 HTTP loopback 串走每次都得 TCP + 鉴权中间件 + Pydantic 反序列化一遍。每个
|
||
# 调用 10-30ms,加起来对前端就是「点进 Step 2 后席位灯一颗一颗亮」的卡顿来源。
|
||
#
|
||
# 这里把这三个调用改成直接吃 app.state 里的 store/checkpointer —— 跑在同一个事
|
||
# 件循环里,省掉 TCP + 序列化 + 中间件链。鉴权 / 校验 / 文件副作用按需在直调
|
||
# helper 里显式做掉,不要"绕过"任何一项。
|
||
#
|
||
# 设计边界:这里只覆盖 init 阶段。SSE 长连接(/runs/stream / state)留在 HTTP
|
||
# loopback,因为它们本就是别的进程模型(LangGraph 子树),抽出来风险大收益小。
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
def _resolve_user_id(request: Request) -> str:
|
||
"""与 routers/agents._current_user_id 完全等价,直调路径里用。"""
|
||
user = getattr(request.state, "user", None)
|
||
if user is not None:
|
||
return str(user.id)
|
||
return get_effective_user_id()
|
||
|
||
|
||
async def _create_thread_direct(request: Request, *, metadata: dict[str, Any] | None = None) -> str:
|
||
"""同进程版的 ``POST /api/threads``。
|
||
|
||
完全镜像 ``routers/threads.create_thread`` 的写盘动作(thread_meta +
|
||
空 checkpoint),但跳过 TCP + Pydantic 反序列化。仅供 init 阶段批量建席位
|
||
线程用,失败时抛 HTTPException 由 SSE 帧上报。
|
||
|
||
``metadata`` 写入 thread 元数据;init 传 ``_ROUNDTABLE_THREAD_METADATA`` 把
|
||
席位 / 协调线程标记成系统会话,从前端「最近的对话」里隐藏。
|
||
"""
|
||
from app.gateway.deps import get_checkpointer, get_thread_store
|
||
|
||
checkpointer = get_checkpointer(request)
|
||
thread_store = get_thread_store(request)
|
||
thread_id = str(uuid.uuid4())
|
||
now = now_iso()
|
||
|
||
try:
|
||
await thread_store.create(thread_id, assistant_id=None, metadata=metadata or {})
|
||
except Exception:
|
||
logger.exception("[init-direct] thread_meta create failed for %s", thread_id)
|
||
raise HTTPException(status_code=500, detail="Failed to create thread")
|
||
|
||
config = {"configurable": {"thread_id": thread_id, "checkpoint_ns": ""}}
|
||
ckpt_metadata = {
|
||
"step": -1,
|
||
"source": "input",
|
||
"writes": None,
|
||
"parents": {},
|
||
"created_at": now,
|
||
}
|
||
|
||
async def _cleanup_partial_thread() -> None:
|
||
delete_checkpoint = getattr(checkpointer, "adelete_thread", None)
|
||
if callable(delete_checkpoint):
|
||
try:
|
||
await delete_checkpoint(thread_id)
|
||
except Exception:
|
||
logger.exception("[init-direct] partial checkpoint cleanup failed for %s", thread_id)
|
||
try:
|
||
await thread_store.delete(thread_id)
|
||
except Exception:
|
||
logger.exception("[init-direct] partial thread_meta cleanup failed for %s", thread_id)
|
||
|
||
try:
|
||
await checkpointer.aput(config, empty_checkpoint(), ckpt_metadata, {})
|
||
except asyncio.CancelledError:
|
||
# A sibling seat may fail while this task is between metadata commit and
|
||
# checkpoint creation. Compensate before propagating cancellation, under
|
||
# shield so a second cancellation cannot abort the cleanup half-way.
|
||
await asyncio.shield(_cleanup_partial_thread())
|
||
raise
|
||
except Exception:
|
||
logger.exception("[init-direct] checkpoint create failed for %s", thread_id)
|
||
await _cleanup_partial_thread()
|
||
raise HTTPException(status_code=500, detail="Failed to create thread")
|
||
return thread_id
|
||
|
||
|
||
async def _get_agents_batch_direct(
|
||
request: Request,
|
||
agent_ids: list[str],
|
||
) -> dict[str, dict[str, Any]]:
|
||
"""同进程版的 N 个并行 ``GET /api/agents/{id}`` 合并为 1 次 DB 查询。
|
||
|
||
返回 ``{agent_id: {id, name, description}}``。**严格模式**:任一席位对当前
|
||
用户不可见就抛 404 —— 与原 HTTP loopback 的 ``_get_agent`` 行为一致(原路径
|
||
单查 404 会 raise,在外层被 except 捕获并 yield error 帧),不要悄悄回退。
|
||
"""
|
||
from app.gateway.deps import get_agent_store
|
||
|
||
store = get_agent_store(request)
|
||
user_id = _resolve_user_id(request)
|
||
records = await store.list_by_ids(agent_ids, user_id)
|
||
out: dict[str, dict[str, Any]] = {}
|
||
for r in records:
|
||
rid = str(r.get("id") or "").lower()
|
||
if not rid:
|
||
continue
|
||
out[rid] = {
|
||
"id": rid,
|
||
"name": str(r.get("name") or rid),
|
||
"description": str(r.get("description") or ""),
|
||
}
|
||
missing = [sid for sid in agent_ids if sid.lower() not in out]
|
||
if missing:
|
||
raise HTTPException(
|
||
status_code=404,
|
||
detail=f"Agent(s) not visible to current user: {', '.join(missing)}",
|
||
)
|
||
return out
|
||
|
||
|
||
# NOTE: 历史上这里曾有 _create_coordinator_direct(),每次 init 都创建一个带时间戳的
|
||
# roundtable-coordinator-<ts36>。2026-05 改造为单例(见 _roundtable_seed.py 与本
|
||
# 模块 init_multi_agent 的 docstring),不再需要"每次创建"的直调辅助;改造后由
|
||
# ensure_roundtable_functional_agents() 在路由入口落地静态 SOUL,席位列表通过
|
||
# _append_thread_message 注入到 coordinator thread 的对话历史中。
|
||
|
||
|
||
async def _append_thread_message(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
thread_id: str,
|
||
content: str,
|
||
) -> None:
|
||
"""向指定 thread 的 checkpoint 追加一条 human 消息(广播上下文,不触发 LLM run)。
|
||
|
||
失败只打 warning,不阻断主流程——广播是 best-effort,席位真正干活仍靠
|
||
前端对该席位 thread 发起的 ``/run/stream``。
|
||
|
||
走**轻量追加**接口(``POST /messages/append``):只传一条消息,服务端读 checkpoint+追加+写回,
|
||
省掉原先 GET /state(下载全量)+POST /state(上传全量)把几万字状态经 loopback 传两遍的开销。
|
||
"""
|
||
resp = await client.post(
|
||
f"{base}/api/threads/{thread_id}/messages/append",
|
||
json={"content": content, "type": "human"},
|
||
headers=headers,
|
||
)
|
||
if resp.status_code >= 400:
|
||
logger.warning("append message %s failed: %s %s", thread_id, resp.status_code, resp.text[:200])
|
||
|
||
|
||
def _spawn_detached_broadcast(
|
||
base: str,
|
||
headers: dict[str, str],
|
||
message: str,
|
||
agent_threads: dict[str, str],
|
||
skip: list[str],
|
||
) -> None:
|
||
"""后台 fire-and-forget 广播,**用独立 client**,即使发起它的请求 SSE 已结束也能跑完。
|
||
|
||
广播是 best-effort 的「上下文播报」(告知其它席位某人交付了什么),原本在总控 yield 派活
|
||
帧**之前**被 ``await``,导致每个席位都要等几十秒(广播要读写各席位的巨大 thread 状态)
|
||
才开始执行 —— 实测 ``dispatch_broadcasts_await`` 飙到 27s。改成 detached 后台任务后,总控
|
||
立即 yield 派活、前端立即启动席位,广播在后台慢慢落地,彻底移出关键路径。失败只记日志。
|
||
"""
|
||
async def _run() -> None:
|
||
try:
|
||
async with httpx.AsyncClient(timeout=None, trust_env=False) as bg:
|
||
await _broadcast(bg, base, headers, message, agent_threads, skip)
|
||
except Exception as exc: # noqa: BLE001
|
||
logger.warning("detached broadcast failed: %s", exc)
|
||
|
||
# create_task 后不持有引用:任务挂在事件循环上,独立于本请求生命周期跑完。
|
||
asyncio.ensure_future(_run())
|
||
|
||
|
||
async def _broadcast(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
message: str,
|
||
agent_threads: dict[str, str],
|
||
skip: list[str],
|
||
) -> None:
|
||
"""并行向 ``agent_threads`` 中除 ``skip`` 外的所有 thread 追加同一条 human 消息。"""
|
||
targets = [tid for name, tid in agent_threads.items() if name not in skip]
|
||
if not targets:
|
||
return
|
||
await asyncio.gather(
|
||
*(_append_thread_message(client, base, headers, tid, message) for tid in targets),
|
||
return_exceptions=True,
|
||
)
|
||
|
||
|
||
_TOOL_CALL_PATTERN = re.compile(r'\.py\s+"([^"]*)"\s+"([^"]*)"')
|
||
|
||
# 子智能体交付的「后广播」里,正文广播给**其它席位**时的截断上限。
|
||
# 背景:席位交付动辄几万字;原实现把整篇交付广播进**每一个**其它席位 thread,导致
|
||
# 每个 thread 状态滚雪球,之后每次状态读/写(广播、交付状态核对)都要搬运巨量数据,
|
||
# 延迟复利式恶化(实测 dispatch_broadcasts_await 飙到 30s)。其它席位只需"知道某席位
|
||
# 交付了什么"的摘要即可;**完整交付仍原样发给总控 thread**(总控做最终综合要看全文),
|
||
# 也仍通过 SSE 完整给到前端。仅截断"广播给兄弟席位"这一路。
|
||
_BROADCAST_CONTENT_MAX_CHARS = 1500
|
||
|
||
|
||
def _truncate_broadcast_content(prefix: str, content: str) -> str:
|
||
"""构造广播给**兄弟席位**的交付摘要消息:超长正文截断到上限并标注省略。
|
||
|
||
``content`` 未超限时原样拼接(等价于完整版),超限时截断 + 追加省略说明。
|
||
总控那一路不走本函数,始终用完整正文(见 ``_special_run`` 的 ``full_msg``)。
|
||
"""
|
||
if len(content) <= _BROADCAST_CONTENT_MAX_CHARS:
|
||
return prefix + content
|
||
return (
|
||
prefix
|
||
+ content[:_BROADCAST_CONTENT_MAX_CHARS]
|
||
+ f"\n…(完整交付约 {len(content)} 字,已省略;以该席位自身交付为准)"
|
||
)
|
||
|
||
# Mirrors the upstream agent-id validator in the LangGraph runtime. Reject early
|
||
# so callers get a clear 400 here instead of a stream that aborts with HTTP 422
|
||
# from the loopback /api/threads/.../runs/stream endpoint.
|
||
_AGENT_ID_PATTERN = re.compile(r"^[A-Za-z0-9_-]+$")
|
||
|
||
|
||
def _validate_agent_id(value: str, field: str) -> str:
|
||
if not _AGENT_ID_PATTERN.match(value):
|
||
raise HTTPException(
|
||
status_code=400,
|
||
detail=(
|
||
f"{field} '{value}' is not a valid agent id. "
|
||
f"Must match ^[A-Za-z0-9_-]+$ (use the hyphen-case id, not the display name)."
|
||
),
|
||
)
|
||
return value
|
||
|
||
|
||
def _extract_dispatch(tool_call: dict[str, Any]) -> tuple[str, str]:
|
||
"""Return ``(agent_name, task)`` parsed from a leader tool call.
|
||
|
||
Two formats are accepted:
|
||
|
||
1. **Structured** — ``{"name": "agent_orchestration", "args": {"agent_name": ..., "task": ...}}``.
|
||
This is what most LLMs emit when told to "use the agent_orchestration skill",
|
||
and the form the roundtable coordinator's SOUL prompt elicits.
|
||
|
||
2. **Legacy bash argv** — ``{"name": "bash", "args": {"command": '... .py "X" "Y" ...'}}``.
|
||
Kept as a fallback so older skills that dispatch through bash still work.
|
||
"""
|
||
name = (tool_call.get("name") or "").lower()
|
||
args = tool_call.get("args")
|
||
|
||
# Format 1: structured agent_orchestration call.
|
||
if name == "agent_orchestration" and isinstance(args, dict):
|
||
agent_name = str(args.get("agent_name") or "").strip()
|
||
task = str(args.get("task") or "").strip()
|
||
if agent_name and task:
|
||
return _strip_display_suffix(agent_name), task
|
||
|
||
# Format 2: legacy bash + .py "X" "Y".
|
||
match = _TOOL_CALL_PATTERN.search(str(args))
|
||
if match:
|
||
return _strip_display_suffix(match.group(1)), match.group(2)
|
||
|
||
return "", ""
|
||
|
||
|
||
def _strip_display_suffix(agent_name: str) -> str:
|
||
"""Tolerate models that paste the human-readable suffix into agent_name.
|
||
|
||
The coordinator SOUL lists each seat as ``agent_name: roundtable-x(中文):...``
|
||
so weaker models sometimes echo the entire ``roundtable-x(中文)`` label
|
||
back as the dispatch target. Strip everything from the first non-id
|
||
character onward so we recover the bare hyphen-case id; if what's left
|
||
still violates the id pattern we hand it back untouched and let the
|
||
upstream validator reject it.
|
||
"""
|
||
head = re.split(r"[^A-Za-z0-9_-]", agent_name, maxsplit=1)[0]
|
||
return head or agent_name
|
||
|
||
|
||
def _flatten_content(content: Any) -> str:
|
||
if isinstance(content, str):
|
||
return content
|
||
if isinstance(content, list):
|
||
parts: list[str] = []
|
||
for item in content:
|
||
if isinstance(item, dict):
|
||
parts.append(str(item.get("text", "")))
|
||
else:
|
||
parts.append(str(item))
|
||
return "".join(parts)
|
||
return str(content)
|
||
|
||
|
||
def _last_ai_raw_text(messages: list[Any]) -> str:
|
||
"""最后一条有原文的 AI 消息(含思考标签),用于区分 thinking_only 与 empty。"""
|
||
for msg in reversed(messages or []):
|
||
if not isinstance(msg, dict):
|
||
continue
|
||
if (msg.get("type") or "").lower() not in ("ai", "aimessage", "aimessagechunk"):
|
||
continue
|
||
raw = _flatten_content(msg.get("content", ""))
|
||
if raw.strip():
|
||
return raw
|
||
return ""
|
||
|
||
|
||
def _visible_delivery_from_messages(messages: list[Any]) -> str:
|
||
"""本轮可见正文:最后一条 human 之后、剥掉思考仍非空的 AI 文本。"""
|
||
last_human = -1
|
||
for i, msg in enumerate(messages or []):
|
||
if isinstance(msg, dict) and (msg.get("type") or "").lower() in ("human", "user"):
|
||
last_human = i
|
||
for msg in reversed((messages or [])[last_human + 1 :]):
|
||
if not isinstance(msg, dict):
|
||
continue
|
||
if (msg.get("type") or "").lower() not in ("ai", "aimessage", "aimessagechunk"):
|
||
continue
|
||
visible = visible_seat_text(_flatten_content(msg.get("content", "")))
|
||
if visible:
|
||
return visible
|
||
return ""
|
||
|
||
|
||
def _run_payload(
|
||
*,
|
||
agent_name: str,
|
||
model_name: str,
|
||
thread_id: str,
|
||
new_message: str,
|
||
excluded_tools: list[str],
|
||
skill_stop_names: list[str],
|
||
thinking_enabled: bool = True,
|
||
reasoning_effort: str = "medium",
|
||
subagent_enabled: bool = False,
|
||
stream_custom: bool = False,
|
||
multitask_strategy: str = "reject",
|
||
extra_context: dict[str, Any] | None = None,
|
||
files: list[dict[str, Any]] | None = None,
|
||
) -> dict[str, Any]:
|
||
"""构造 ``POST /api/threads/{thread_id}/runs/stream`` 的请求体。
|
||
|
||
- ``context.agent_name``:本次 run 使用的 agent(内置 id 或 init 创建的 coordinator id)。
|
||
- ``excluded_tools``:按路由裁剪工具面(intent 仅澄清、special 禁派活等)。
|
||
- ``skill_stop_names``:leader 侧配合 SkillStopMiddleware 在 ``agent_orchestration``
|
||
调用后截断图,以便网关从 AIMessage.tool_calls 解析派活列表。
|
||
- ``thinking_enabled`` / ``reasoning_effort``:默认开 CoT + medium,适合 Step 1 澄清
|
||
与子智能体交付这类需要推理的场景。**推荐弹窗与总控派活**是短决策任务,调用方
|
||
应传 ``thinking_enabled=False`` + ``reasoning_effort="low"``,省掉模型先吐
|
||
reasoning tokens 才开口的时间,显著降低前端"调度中"首字节延迟。
|
||
"""
|
||
# 用户上传的文件元数据透传给 UploadsMiddleware(与普通聊天同链路):
|
||
# before_agent 钩子读 additional_kwargs.files,生成 <uploaded_files> 清单注入消息,
|
||
# 之后席位/总控用内置 read_file / grep / view_image 工具按需读取。空时落空 dict,行为不变。
|
||
additional_kwargs: dict[str, Any] = {}
|
||
if files:
|
||
additional_kwargs["files"] = files
|
||
payload: dict[str, Any] = {
|
||
"input": {
|
||
"messages": [
|
||
{
|
||
"type": "human",
|
||
"content": [{"type": "text", "text": new_message}],
|
||
"additional_kwargs": additional_kwargs,
|
||
}
|
||
]
|
||
},
|
||
"metadata": {"run_scope": RUN_SCOPE_ORCHESTRATION_CHILD},
|
||
"config": {"recursion_limit": 1000},
|
||
"context": {
|
||
"agent_name": agent_name,
|
||
"model_name": model_name,
|
||
"mode": "pro",
|
||
"reasoning_effort": reasoning_effort,
|
||
"thinking_enabled": thinking_enabled,
|
||
"is_plan_mode": False,
|
||
# 默认 False;圆桌「多智能体问答(ultra)」档会让席位 _special_run 传 True,
|
||
# 给席位解锁 task 子代理工具(席位 policy 未禁 task,故可用)。
|
||
"subagent_enabled": subagent_enabled,
|
||
"thread_id": thread_id,
|
||
# 圆桌四个 router(intent / recommend / leader / special)都是临时任务,
|
||
# 用户个人记忆与本次研讨无关。MemoryMiddleware 的 Hindsight recall 是
|
||
# 同步阻塞 HTTP 调用(单次 timeout 15s),如果记忆服务慢/不通会让前端
|
||
# "调度中"卡很久;这里默认跳过。
|
||
"memory_recall_disabled": True,
|
||
# ⚠️ 同时关掉 builtin 记忆**注入**:圆桌四个角色(intent / recommend /
|
||
# leader / special)都是固定单例 agent,其 memory.json 跨所有会话累积。
|
||
# 总控尤其危险——上一个任务里它臆造的席位名(如 data-specialist)会被存进
|
||
# memory.json,下一个新任务又被注入系统提示而召回 → 派给根本没初始化的
|
||
# 席位 → run/stream 报 400 "agent_name ... missing from d_agent_thread_id"。
|
||
# 圆桌是一次性研讨,本就不该吃历史会话记忆,这里统一关注入根治该污染。
|
||
"memory_injection_enabled": False,
|
||
},
|
||
# ultra 席位会派子智能体(task 工具),其 task_started/running/completed 进度走
|
||
# **custom** 流(StreamWriter)。仅这种场景加 "custom",让前端任务卡片能显示子智能体
|
||
# 实时进度;其余角色不加,避免无谓帧。
|
||
"stream_mode": (
|
||
["messages-tuple", "values", "custom"] if stream_custom else ["messages-tuple", "values"]
|
||
),
|
||
"stream_subgraphs": True,
|
||
"stream_resumable": True,
|
||
"assistant_id": "lead_agent",
|
||
"on_disconnect": "continue",
|
||
# ⚠️ 圆桌 run 必须能被「下一次同 thread 的 run」抢占。圆桌每个 thread 同一时刻
|
||
# 只应有一个活跃 run,但 on_disconnect="continue" 让前端 abort SSE 后**后端 run
|
||
# 仍在跑**(为可恢复);此时用户「人工干预」立刻发起新 leader run 打到同一总控
|
||
# thread,默认 reject 策略就会 409「already has an active run」→ 被 _stream_upstream
|
||
# 包成 502。调用方(leader / special)传 multitask_strategy="rollback",让新 run 先
|
||
# 取消该 thread 上没结束的旧 run 再继续 —— 这正是「最新一轮/干预指令优先」的语义。
|
||
"multitask_strategy": multitask_strategy,
|
||
"excluded_tools": excluded_tools,
|
||
"skill_stop_names": skill_stop_names,
|
||
}
|
||
# 额外注入 run context(如圆桌席位的 roundtable_peer_deliveries:前序席位完整交付,
|
||
# 供 read_peer_delivery 工具按需读取)。只进 context dict,不进 prompt → 不调工具就不耗 token。
|
||
if extra_context:
|
||
payload["context"].update(extra_context)
|
||
return payload
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# POST /api/multi-agent/init — Step 2 席位 + 总控初始化(SSE)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
@router.post("/init")
|
||
async def init_multi_agent(request: Request, payload: dict = Body(...)) -> StreamingResponse:
|
||
"""流式初始化圆桌 Step 2 所需的 thread 映射与协调智能体。
|
||
|
||
请求体::
|
||
|
||
{
|
||
"agent_names": ["roundtable-intelligence", ...], // 用户选的席位 id
|
||
"main_agent_name": "<可选,已废弃>", // 历史兼容字段,新代码忽略,统一用 COORDINATOR_AGENT_ID
|
||
"model": "<可选>" // 可选;不传由 lead_agent 回退到 config.yaml models[0]
|
||
}
|
||
|
||
SSE 事件(每条 ``data: {...}\\n\\n``)::
|
||
|
||
seat_ready — 某一席位 thread 已创建(``asyncio.as_completed`` 顺序,先完先发)
|
||
coordinator_ready — 总控 thread 就绪 + 已注入本次席位列表
|
||
init_done — ``thread_ids`` 全量映射,前端存 ``threadIds`` 并开始编排
|
||
error — 失败详情(同步校验失败则直接 HTTP 4xx,不进 SSE)
|
||
|
||
协调智能体改造说明
|
||
----------------
|
||
协调智能体现在是 **单例**(id 固定为 ``roundtable-coordinator``,SOUL 见
|
||
``_roundtable_seed.COORDINATOR_AGENT_SOUL``),由 ``ensure_roundtable_functional_agents()``
|
||
在首次调用时自动落地。每次 init **不再创建新的 coordinator agent**(避免
|
||
``.deer-flow/agents/`` 累积 ``roundtable-coordinator-<ts36>`` 残留)。
|
||
|
||
本次圆桌的「可调度席位列表」由 init 在创建好 coordinator thread 之后,以一条
|
||
human 消息追加到 thread 历史里,leader run 每轮都能从对话历史里读到。这样
|
||
SOUL 长期稳定,席位列表随会话动态注入。前端为兼容旧 client 仍可传
|
||
``main_agent_name`` 字段,但后端会忽略并使用 ``COORDINATOR_AGENT_ID``。
|
||
"""
|
||
# 调用前自检:与 intent/recommend 一样,缺失时就地补建,避免 /run/stream 报 500。
|
||
ensure_roundtable_functional_agents()
|
||
|
||
agent_names_in = payload.get("agent_names") or []
|
||
agent_names: list[str] = [str(name).lower() for name in agent_names_in]
|
||
model: str | None = payload.get("model")
|
||
# main_agent_name 是历史字段,2026-05 改造后忽略,统一用单例 id。保留接收但不再用前端值,
|
||
# 避免下游再次创建带时间戳的临时 coordinator。
|
||
main_agent_name = COORDINATOR_AGENT_ID
|
||
|
||
if not agent_names:
|
||
raise HTTPException(status_code=400, detail="agent_names is required")
|
||
|
||
for name in agent_names:
|
||
_validate_agent_id(name, "agent_names entry")
|
||
|
||
def _frame(payload: dict[str, Any]) -> str:
|
||
return f"data: {json.dumps(payload, ensure_ascii=False)}\n\n"
|
||
|
||
async def stream_generator() -> AsyncIterator[str]:
|
||
# init 阶段 fast path:全部走同进程直调(_create_thread_direct /
|
||
# _get_agents_batch_direct / _create_coordinator_direct),不再走 HTTP
|
||
# loopback —— TCP + 鉴权中间件 + Pydantic 反序列化每次 10-30ms,N+1 个席位
|
||
# 累计就是"点进 Step 2 后席位一颗颗亮"的卡顿来源。
|
||
try:
|
||
t_init_start = time.perf_counter()
|
||
# 1) 一次 DB 查询拿到所有席位的 name/description,替代之前的 N 个并行
|
||
# GET /api/agents/{id}。SOUL 拼接需要描述,这里一次性拿齐。
|
||
agent_meta = await _get_agents_batch_direct(request, agent_names)
|
||
t_meta_done = time.perf_counter()
|
||
|
||
# 2) 每个席位 thread 并行创建(直调 thread_store + checkpointer),
|
||
# 用 asyncio.as_completed 保持"先完先发 seat_ready"的体验。
|
||
async def _prepare_sub_agent(sub_id: str) -> tuple[str, str, str, str]:
|
||
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
meta = agent_meta.get(sub_id.lower(), {})
|
||
return (
|
||
sub_id.lower(),
|
||
thread_id,
|
||
str(meta.get("name") or sub_id),
|
||
str(meta.get("description") or ""),
|
||
)
|
||
|
||
tasks = [asyncio.create_task(_prepare_sub_agent(sid)) for sid in agent_names]
|
||
prepared: list[tuple[str, str, str, str]] = []
|
||
try:
|
||
for fut in asyncio.as_completed(tasks):
|
||
sid, thread_id, display_name, description = await fut
|
||
prepared.append((sid, thread_id, display_name, description))
|
||
yield _frame({
|
||
"event": "seat_ready",
|
||
"agent_name": sid,
|
||
"thread_id": thread_id,
|
||
"display_name": display_name,
|
||
"description": description,
|
||
})
|
||
except Exception:
|
||
# Cancel any still-pending sibling so partially-prepared seats
|
||
# don't keep running past the stream close.
|
||
for t in tasks:
|
||
if not t.done():
|
||
t.cancel()
|
||
raise
|
||
t_seats_done = time.perf_counter()
|
||
|
||
agent_threads: dict[str, str] = {sid: thread_id for sid, thread_id, _, _ in prepared}
|
||
seat_lines: list[str] = [
|
||
f"- agent_name: {sid}({display_name}):{description}"
|
||
for sid, _, display_name, description in prepared
|
||
]
|
||
|
||
# 协调智能体改造为**单例**(id = roundtable-coordinator):
|
||
# - SOUL 已经由 ensure_roundtable_functional_agents() 在路由入口落地,
|
||
# 通用协调规则常驻磁盘,无需每次重写。
|
||
# - 本次"可调度席位清单"以 human 消息形式 append 到 coordinator
|
||
# thread 历史里,leader run 每轮都能在对话历史最前面看到。
|
||
# 这样席位列表随会话动态变化,但磁盘上不再累积 roundtable-coordinator-<ts36>。
|
||
#
|
||
# 注意:model 字段在 lead_agent run 时通过 context.model_name 注入,
|
||
# 不需要写回到 agent config.yaml(否则会污染单例的配置)。
|
||
seat_introduction = (
|
||
"【圆桌系统初始化】本次研讨可调度的席位清单如下,请严格按 agent_name "
|
||
"通过 agent_orchestration 技能派活,不要派给不在清单内的 agent:\n"
|
||
+ "\n".join(seat_lines)
|
||
)
|
||
|
||
coord_thread = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
agent_threads[main_agent_name] = coord_thread
|
||
|
||
# 把席位列表注入 coordinator thread 的对话历史。HTTP loopback 调用
|
||
# 走 /api/threads/{tid}/state(读+写),开销 ~10-30ms,在 init 整体
|
||
# 几百 ms 的尺度上微不足道,但比起手撕 langgraph checkpoint 内部结构
|
||
# 风险低、与 _broadcast / _append_thread_message 一套语义,长期可维护。
|
||
try:
|
||
base = _loopback_base(request)
|
||
headers = _auth_headers(request)
|
||
async with _make_loopback_client(request) as seed_client:
|
||
await _append_thread_message(
|
||
seed_client,
|
||
base,
|
||
headers,
|
||
coord_thread,
|
||
seat_introduction,
|
||
)
|
||
except Exception:
|
||
# 注入失败不阻断 init:leader run 仍能跑,只是看不到席位清单,
|
||
# 可能派活给错的 id。会被 _strip_display_suffix / _validate_agent_id
|
||
# 兜底,前端能看到错误帧;运维通过 backend 日志找到根因再修复。
|
||
logger.exception("[init-direct] failed to seed coordinator thread with seat list")
|
||
|
||
t_coord_done = time.perf_counter()
|
||
logger.info(
|
||
"[init-direct] meta=%.3fs seats=%.3fs coord=%.3fs total=%.3fs seats_count=%d",
|
||
t_meta_done - t_init_start,
|
||
t_seats_done - t_meta_done,
|
||
t_coord_done - t_seats_done,
|
||
t_coord_done - t_init_start,
|
||
len(prepared),
|
||
)
|
||
|
||
yield _frame({
|
||
"event": "coordinator_ready",
|
||
"main_agent_name": main_agent_name,
|
||
"thread_id": coord_thread,
|
||
})
|
||
yield _frame({
|
||
"event": "init_done",
|
||
"main_agent_name": main_agent_name,
|
||
"thread_ids": agent_threads,
|
||
})
|
||
except HTTPException as exc:
|
||
logger.warning("multi-agent init HTTPException: %s %s", exc.status_code, exc.detail)
|
||
await record_foreground_diag(
|
||
request, stage="step2_init", level="error", event="init_http_error",
|
||
message=f"会商初始化失败 [{exc.status_code}]:{exc.detail}",
|
||
detail={"status_code": exc.status_code, "detail": str(exc.detail)},
|
||
)
|
||
yield _frame({"event": "error", "detail": f"[{exc.status_code}] {exc.detail}"})
|
||
except Exception as exc:
|
||
logger.exception("multi-agent init unexpected failure")
|
||
await record_foreground_diag(
|
||
request, stage="step2_init", level="error", event="init_unexpected",
|
||
message=f"会商初始化未预期异常:{exc!r}",
|
||
detail={"type": type(exc).__name__, "error": str(exc)},
|
||
)
|
||
yield _frame({"event": "error", "detail": f"unexpected: {exc!r}"})
|
||
|
||
return StreamingResponse(
|
||
stream_generator(),
|
||
media_type="text/event-stream",
|
||
headers={
|
||
"Cache-Control": "no-cache",
|
||
"X-Accel-Buffering": "no",
|
||
"Connection": "keep-alive",
|
||
},
|
||
)
|
||
|
||
|
||
@router.post("/report/init")
|
||
async def init_report_thread(request: Request) -> dict[str, str]:
|
||
"""为 Step 3「结果绘制」创建一个供方案可视化总结智能体(``roundtable-report``)
|
||
运行的线程,返回 ``{"agent_name", "thread_id"}``。
|
||
|
||
设计要点:
|
||
- ``roundtable-report`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进
|
||
DB),这里**只建一个轻量 thread**,绝不重建 agent;
|
||
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
|
||
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
|
||
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
|
||
``agent_name="roundtable-report"``、``d_agent_thread_id={REPORT_AGENT_ID: thread_id}``)
|
||
跑这一轮,智能体把图表 HTML 写进该线程的 ``outputs/``,再由沙箱预览。
|
||
"""
|
||
try:
|
||
ensure_roundtable_functional_agents()
|
||
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
except Exception as exc:
|
||
await record_foreground_diag(
|
||
request, stage="step3_report", level="error", event="thread_init_failed",
|
||
message=f"创建结果绘制线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
|
||
agent_id=REPORT_AGENT_ID,
|
||
)
|
||
raise
|
||
return {"agent_name": REPORT_AGENT_ID, "thread_id": thread_id}
|
||
|
||
|
||
@router.post("/summary/init")
|
||
async def init_summary_thread(request: Request) -> dict[str, str]:
|
||
"""为 Step 3「总结报告」创建一个供方案总结报告智能体(``roundtable-summary``)
|
||
运行的线程,返回 ``{"agent_name", "thread_id"}``。
|
||
|
||
与 ``/report/init`` 同构,只是换成另一个内置单例 ``roundtable-summary``:
|
||
- ``roundtable-summary`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进
|
||
DB),这里**只建一个轻量 thread**,绝不重建 agent;
|
||
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
|
||
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
|
||
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
|
||
``agent_name="roundtable-summary"``、``d_agent_thread_id={SUMMARY_AGENT_ID: thread_id}``)
|
||
跑这一轮(后续问答/增量改**复用同一线程**,不重新 init),智能体把 Markdown 报告
|
||
写进该线程的 ``outputs/方案总结报告.md``,再由沙箱预览。
|
||
"""
|
||
try:
|
||
ensure_roundtable_functional_agents()
|
||
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
except Exception as exc:
|
||
await record_foreground_diag(
|
||
request, stage="step3_summary", level="error", event="thread_init_failed",
|
||
message=f"创建总结报告线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
|
||
agent_id=SUMMARY_AGENT_ID,
|
||
)
|
||
raise
|
||
return {"agent_name": SUMMARY_AGENT_ID, "thread_id": thread_id}
|
||
|
||
|
||
@router.post("/position-action-plan/init")
|
||
async def init_position_action_plan_thread(request: Request) -> dict[str, str]:
|
||
"""Create the reusable thread for the position-roundtable action planner.
|
||
|
||
The planner is a built-in singleton agent just like ``roundtable-summary``;
|
||
only its thread is per session. The position workspace persists this id
|
||
after the first successful run so follow-up revisions retain the original
|
||
plan context.
|
||
"""
|
||
try:
|
||
ensure_roundtable_functional_agents()
|
||
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
except Exception as exc:
|
||
await record_foreground_diag(
|
||
request,
|
||
stage="position_action_plan",
|
||
level="error",
|
||
event="thread_init_failed",
|
||
message=f"创建岗位行动规划线程失败:{exc}",
|
||
detail={"type": type(exc).__name__, "error": str(exc)},
|
||
agent_id=ACTION_PLAN_AGENT_ID,
|
||
)
|
||
raise
|
||
return {"agent_name": ACTION_PLAN_AGENT_ID, "thread_id": thread_id}
|
||
|
||
|
||
@router.post("/dashboard/init")
|
||
async def init_dashboard_thread(request: Request) -> dict[str, str]:
|
||
"""为 Step 3「大屏多页」创建一个供大屏多页数据智能体(``roundtable-dashboard``)
|
||
运行的线程,返回 ``{"agent_name", "thread_id"}``。
|
||
|
||
与 ``/report/init`` / ``/summary/init`` 同构,换成第三个内置单例
|
||
``roundtable-dashboard``(专职「逐席位分析 → report-json」,run policy 为 ``dashboard``,
|
||
**禁用一切工具**):
|
||
- ``roundtable-dashboard`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进 DB),
|
||
这里**只建一个轻量 thread**,绝不重建 agent;
|
||
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
|
||
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
|
||
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
|
||
``agent_name="roundtable-dashboard"``、``d_agent_thread_id={DASHBOARD_AGENT_ID: thread_id}``)
|
||
跑这一轮(后续问答/增量改**复用同一线程**,不重新 init),智能体在对话里输出
|
||
```report-json,前端用 ``extractReportData`` 抽取后按当前主题渲染大屏。
|
||
"""
|
||
try:
|
||
ensure_roundtable_functional_agents()
|
||
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
except Exception as exc:
|
||
await record_foreground_diag(
|
||
request, stage="step3_dashboard", level="error", event="thread_init_failed",
|
||
message=f"创建大屏数据线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
|
||
agent_id=DASHBOARD_AGENT_ID,
|
||
)
|
||
raise
|
||
return {"agent_name": DASHBOARD_AGENT_ID, "thread_id": thread_id}
|
||
|
||
|
||
@router.post("/structure/init")
|
||
async def init_structure_thread(request: Request) -> dict[str, str]:
|
||
"""为 Step 3「入库」创建一个供结构化抽取智能体(``roundtable-structure``)运行的线程,
|
||
返回 ``{"agent_name", "thread_id"}``。
|
||
|
||
与 ``/report/init`` / ``/summary/init`` / ``/dashboard/init`` 同构,换成内置单例
|
||
``roundtable-structure``。区别在于本轮**不是抽流程图**,而是让它**真正调用入库技能**
|
||
(run policy 为 ``ingest``——放开 ``read_file`` / ``bash`` / ``write_file`` 让技能能执行):
|
||
- ``roundtable-structure`` 是**内置单例**(``_roundtable_seed`` 种子化,启动即同步进 DB),
|
||
这里**只建一个轻量 thread**,绝不重建 agent;
|
||
- 线程打 ``_ROUNDTABLE_THREAD_METADATA`` 标记成系统会话,从「最近对话」里隐藏;
|
||
- 入口先 ``ensure_roundtable_functional_agents()`` 自愈(磁盘目录被手删也能补建);
|
||
- 建好后前端用现有 ``POST /run/stream``(``agent_type="special"``、
|
||
``agent_name="roundtable-structure"``、``d_agent_thread_id={STRUCTURE_AGENT_ID: thread_id}``)
|
||
跑这一轮,把 taskId + 总结报告 + 结构化流程图交给它,由它挑选合适的入库技能(名称
|
||
可能含「六步法」/自行判断适用性)读 SKILL.md 并按其契约调用完成入库。
|
||
"""
|
||
try:
|
||
ensure_roundtable_functional_agents()
|
||
thread_id = await _create_thread_direct(request, metadata=_ROUNDTABLE_THREAD_METADATA)
|
||
except Exception as exc:
|
||
await record_foreground_diag(
|
||
request, stage="step3_structure", level="error", event="thread_init_failed",
|
||
message=f"创建数据入库线程失败:{exc}", detail={"type": type(exc).__name__, "error": str(exc)},
|
||
agent_id=STRUCTURE_AGENT_ID,
|
||
)
|
||
raise
|
||
return {"agent_name": STRUCTURE_AGENT_ID, "thread_id": thread_id}
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# 上游 LangGraph SSE 透传(intent / recommend / leader / special 共用)
|
||
# ---------------------------------------------------------------------------
|
||
|
||
# 回环 /runs/stream 在**响应头阶段**返回这些状态码时,视为「单 worker + 单 sqlite 锁被
|
||
# 并发打满」造成的瞬时失败,退避重试(详见 _stream_upstream 内注释):
|
||
# 429 限流;500/503/504 高并发下 start_run 里 DB/checkpoint 写抖动。
|
||
# 不含 400/401/403/404/422 等确定性错误——那些重试也没用,立即上报。
|
||
#
|
||
# ⚠️ 409 **不在**重试集合里。409 = 同 thread 已有活跃 run(``multitask_strategy=reject``
|
||
# 拒绝)。此时重发**不是**幂等的:LangGraph 会把同一条 human message 再追加一次 → 用户
|
||
# 这一轮被处理两遍(重复澄清卡 / 双发意图),或两次退避后仍 502 —— 这正是用户报的
|
||
# 「确定意图接口报错」。故 409 必须透传:由上层(intent_turns 去重 / 前端按 client_turn_id
|
||
# 等待)决定是等待还是放弃,而不是盲目重发。其余(rollback)调用方本就靠抢占取消在跑 run,
|
||
# 几乎不触 409,移除亦无碍。
|
||
_UPSTREAM_RETRYABLE_STATUSES = frozenset({429, 500, 502, 503, 504})
|
||
# 重试次数(不含首次)。总尝试 = 1 + _UPSTREAM_MAX_RETRIES。
|
||
_UPSTREAM_MAX_RETRIES = 2
|
||
# 指数退避基数(秒):0.4 → 0.8 → …,足够让在跑的席位 run 让出 sqlite 锁,又不至于明显拖慢。
|
||
_UPSTREAM_RETRY_BASE_DELAY = 0.4
|
||
|
||
|
||
async def _stream_upstream(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
thread_id: str,
|
||
body: dict[str, Any],
|
||
) -> AsyncIterator[tuple[str | None, list[str] | None]]:
|
||
"""消费 ``/api/threads/{thread_id}/runs/stream``,逐行 yield 原始 SSE。
|
||
|
||
产出元组:
|
||
- ``(line, None)`` — 可原样转发给浏览器(含 ``event:`` / ``data:`` 行);
|
||
- ``(None, captured)`` — 流结束,``captured`` 为所有 ``data:`` 行文本列表,
|
||
供 ``_parse_last_messages`` 从倒数 values 快照提取 ``messages``。
|
||
|
||
日志 ``[upstream-timing]``:排查首 token 慢;headers_in 长尾若 >> 网络往返(loopback
|
||
本机往返通常 <10ms),怀疑 HTTP_PROXY 拦了回环——检查 client 是否走 _make_loopback_client
|
||
(trust_env=False)。
|
||
"""
|
||
|
||
url = f"{base}/api/threads/{thread_id}/runs/stream"
|
||
captured: list[str] = []
|
||
agent_for_log = (body.get("context") or {}).get("agent_name") or thread_id
|
||
|
||
# 头阶段非 200 的瞬时重试。
|
||
# 背景:网关是**单 worker** + checkpointer 是**单连接 aiosqlite**(全进程一把
|
||
# asyncio.Lock 串行 checkpoint 读写)。圆桌第三步进入时,前端把「方案总结/结果绘制/
|
||
# 大屏」三个重 run **同一刻并行**打出来,叠加第二步 on_disconnect=continue 仍在后台
|
||
# 跑的席位 run,瞬时把单锁/单 worker 打满 —— 此时回环子请求 /runs/stream 的 start_run
|
||
# 里那串 DB / checkpoint 写有概率抛错(或同 thread 抢占竞态),返回非 200,被本函数原样
|
||
# 包成 502 漏给前端(用户侧表现为"进第三步概率性 502")。
|
||
# 关键安全前提:这个非 200 发生在**响应头阶段**,本函数**尚未 yield 任何 SSE 行**给下游,
|
||
# 所以"换条连接重发"是幂等安全的(不会重复透传半截流)。对可恢复的瞬时状态码退避重试几次,
|
||
# 把绝大多数并发尖峰造成的偶发失败吃掉;非可恢复状态码(如 400/401/403/404)立即按原逻辑
|
||
# 报错,不浪费重试。一旦进入流式(已 yield 行)就绝不重试。
|
||
last_status = 0
|
||
last_body = ""
|
||
for attempt in range(_UPSTREAM_MAX_RETRIES + 1):
|
||
t_send = time.perf_counter()
|
||
async with client.stream("POST", url, json=body, headers=headers) as resp:
|
||
t_headers = time.perf_counter()
|
||
if resp.status_code != 200:
|
||
last_status = resp.status_code
|
||
last_body = (await resp.aread()).decode("utf-8", "replace")[:1000]
|
||
if resp.status_code in _UPSTREAM_RETRYABLE_STATUSES and attempt < _UPSTREAM_MAX_RETRIES:
|
||
delay = _UPSTREAM_RETRY_BASE_DELAY * (2**attempt)
|
||
logger.warning(
|
||
"[upstream-retry] %s 回环 %s 返回 HTTP %s,%.2fs 后重试(第 %d/%d 次):%s",
|
||
agent_for_log,
|
||
url,
|
||
resp.status_code,
|
||
delay,
|
||
attempt + 1,
|
||
_UPSTREAM_MAX_RETRIES,
|
||
last_body[:200],
|
||
)
|
||
await asyncio.sleep(delay)
|
||
continue # 重连:换条新连接重发同一 body(尚未 yield 任何行,幂等安全)
|
||
raise HTTPException(
|
||
status_code=502,
|
||
detail=f"upstream {url} returned HTTP {resp.status_code}: {last_body}",
|
||
)
|
||
logger.info(
|
||
"[upstream-timing] %s headers_in=%.3fs%s",
|
||
agent_for_log,
|
||
t_headers - t_send,
|
||
f" (retried x{attempt})" if attempt else "",
|
||
)
|
||
first_line = True
|
||
async for line in resp.aiter_lines():
|
||
if not line:
|
||
continue
|
||
if first_line:
|
||
first_line = False
|
||
logger.info(
|
||
"[upstream-timing] %s first_line=%.3fs (LangGraph runtime startup + LLM first byte)",
|
||
agent_for_log,
|
||
time.perf_counter() - t_headers,
|
||
)
|
||
yield line, None
|
||
if line.startswith("data:"):
|
||
captured.append(line)
|
||
yield None, captured
|
||
return
|
||
|
||
# 理论上不可达(循环要么 return、要么 raise);留一条兜底防御。
|
||
raise HTTPException(
|
||
status_code=502,
|
||
detail=f"upstream {url} returned HTTP {last_status}: {last_body}",
|
||
)
|
||
|
||
|
||
def _parse_last_messages(captured: list[str]) -> list[dict[str, Any]] | None:
|
||
"""Return the full ``messages`` list from the most recent ``values`` snapshot.
|
||
|
||
The upstream sends the final ``values`` frame second-to-last (the very last
|
||
line is the ``end`` event with ``null``). We tolerate trailing non-JSON
|
||
frames (e.g. ``data: null``) by scanning backwards for the first frame that
|
||
successfully parses as a state object with a ``messages`` key.
|
||
|
||
Returns ``None`` when the stream carried **no parseable ``values`` frame**
|
||
(empty capture, or only ``data: null`` / token deltas). 这通常**不是真正的
|
||
上游错误**,而是该 run 被 ``multitask_strategy=rollback`` 抢占 / 客户端断开 /
|
||
中途 abort —— 流被截断,没来得及发出最终 values 快照。调用方(leader/special)据此
|
||
**静默收尾**,绝不能把它当 502 抛给前端:抢占场景下这条 SSE 多半已被前端
|
||
``abortAllInFlight`` 放弃,硬报 502 只会在 UI 上凭空冒出一条错误。
|
||
"""
|
||
if not captured:
|
||
return None
|
||
|
||
for raw_line in reversed(captured):
|
||
raw = raw_line[5:].strip() if raw_line.startswith("data:") else raw_line.strip()
|
||
if not raw or raw == "null":
|
||
continue
|
||
try:
|
||
parsed = json.loads(raw)
|
||
except json.JSONDecodeError:
|
||
continue
|
||
if isinstance(parsed, dict) and "messages" in parsed:
|
||
messages = parsed.get("messages") or []
|
||
if isinstance(messages, list):
|
||
return messages
|
||
|
||
return None
|
||
|
||
|
||
def _find_trailing_ai_text(messages: list[dict[str, Any]], *, exclude_text: str = "") -> str:
|
||
"""Return the text content of the latest AIMessage with no tool_calls.
|
||
|
||
Used by the clarification branch: ClarificationMiddleware ends the run
|
||
after emitting the structured ToolMessage, but the model often produces a
|
||
richer human-readable follow-up AIMessage right before the interrupt.
|
||
Prefer that text over the brief preamble that accompanied the
|
||
ask_clarification tool_call. When the only trailing AIMessage matches the
|
||
preamble (``exclude_text``), return empty so the caller can fall back to
|
||
the question itself.
|
||
"""
|
||
for msg in reversed(messages):
|
||
if not isinstance(msg, dict):
|
||
continue
|
||
if (msg.get("type") or "").lower() not in ("ai", "aimessage", "aimessagechunk"):
|
||
continue
|
||
if msg.get("tool_calls"):
|
||
continue
|
||
text = _flatten_content(msg.get("content", "")).strip()
|
||
if not text:
|
||
continue
|
||
if exclude_text and text == exclude_text.strip():
|
||
continue
|
||
return text
|
||
return ""
|
||
|
||
|
||
def _find_dispatch_message(messages: list[dict[str, Any]]) -> dict[str, Any]:
|
||
"""Locate the latest AIMessage from THIS turn.
|
||
|
||
Trailing ToolMessages (skill_stop / clarification ack injected by
|
||
``SkillStopMiddleware`` or ``ClarificationMiddleware``) are skipped because
|
||
the actionable tool_calls live on the AIMessage that emitted them. We stop
|
||
at the **first AIMessage encountered** — its ``tool_calls`` (or lack
|
||
thereof) is the authoritative signal for this turn.
|
||
|
||
历史踩坑:之前实现是"一直往前翻直到撞到带 tool_calls 的 AI",看似优雅,
|
||
实际上让 leader 永远收敛不了:本轮 LLM 不论怎么共识,只要历史上派过活,
|
||
上一轮的派活 AI 就会被翻出来再用一次,前端会把同样的 task 又发给子智能体,
|
||
陷入死循环。现在停在最新一条 AI 上,把判定交给调用方:`tool_calls=[]` →
|
||
共识分支,有 tool_calls → 派活分支。
|
||
"""
|
||
for msg in reversed(messages):
|
||
if not isinstance(msg, dict):
|
||
continue
|
||
msg_type = (msg.get("type") or "").lower()
|
||
if msg_type in ("tool", "toolmessage"):
|
||
continue # 跳过末尾 skill_stop / clarification ack
|
||
if msg_type in ("ai", "aimessage", "aimessagechunk"):
|
||
return msg
|
||
break # 撞到 Human/其他类型,本轮没 AI(异常路径,留空让上游报错)
|
||
return {}
|
||
|
||
|
||
def _partition_dispatch(
|
||
tool_calls: list[dict[str, Any]],
|
||
valid_seat_ids: set[str],
|
||
) -> tuple[list[tuple[str, str]], list[str]]:
|
||
"""把总控的 agent_orchestration 派活拆成「合法 / 非法」两份。
|
||
|
||
返回 ``(valid, invalid)``:
|
||
|
||
- ``valid`` —— ``[(seat_id, task), ...]``,``seat_id`` 确实是本轮已初始化的席位;
|
||
- ``invalid`` —— ``[seat_id, ...]``,模型臆造 / 不在本轮 roster 内的 id。
|
||
|
||
弱模型(deepseek-chat)会把 SOUL 里的示例 hex id 当真照抄、或臆造同款 32 位 hex,
|
||
这些 id 不在 ``agent_threads`` 里;若不拦截会漏给前端,到 ``/run/stream`` 才报 400
|
||
把整个 Step 2 打挂。校验放在派活解析处,只丢非法、保留同轮合法席位。
|
||
"""
|
||
valid: list[tuple[str, str]] = []
|
||
invalid: list[str] = []
|
||
for tool_call in tool_calls:
|
||
sub_name, task_text = _extract_dispatch(tool_call)
|
||
if not sub_name:
|
||
continue
|
||
if sub_name in valid_seat_ids:
|
||
valid.append((sub_name, task_text))
|
||
else:
|
||
invalid.append(sub_name)
|
||
return valid, invalid
|
||
|
||
|
||
def _build_dispatch_correction(invalid_names: list[str], valid_seat_ids: set[str]) -> str:
|
||
"""整轮派活全非法时,回灌给总控 thread 的系统纠正消息文本。
|
||
|
||
列出真实席位 id,要求模型**逐字符复制**重派,杜绝「凭示例/记忆构造 id」。
|
||
"""
|
||
bad = "、".join(invalid_names)
|
||
roster = "、".join(sorted(valid_seat_ids))
|
||
return (
|
||
"【系统纠正】你刚才通过 agent_orchestration 派活的 agent_name("
|
||
f"{bad})不在本次圆桌的席位清单里,无法执行。请**只**从下面这份真实席位清单里"
|
||
"逐字符复制 agent_name 重新派活,不要凭记忆、示例或猜测自行构造 id:\n"
|
||
f"{roster}"
|
||
)
|
||
|
||
|
||
def _build_seat_roster_block(seats: list[tuple[str, dict[str, Any]]]) -> str:
|
||
"""构造注入到总控**本轮输入最前面**的「真实席位清单」文本。
|
||
|
||
``seats`` 为 ``[(seat_id, {name, description}), ...]``。
|
||
|
||
为什么要每轮把清单塞进输入,而不是只靠 init 时 _append_thread_message 注入历史:
|
||
圆桌席位清单原本只在 init 时以一条 human 消息异步写进总控 thread 历史,总控靠"读
|
||
历史"看到它 —— 这条链路脆弱(best-effort 写盘、跨 thread 持久化),一旦没落地,弱模型
|
||
(deepseek-chat)面对空历史就会**从任务语义臆造**不存在的席位(已观测到 architecture-
|
||
specialist、02714cbe… 这类假 id)。网关在每次 leader run 时本就握有真实席位 id,这里
|
||
直接把清单摆到总控本轮输入最前面并要求逐字复制,从源头根治臆造;与事后 _partition_dispatch
|
||
校验构成「源头 + 兜底」双保险。
|
||
"""
|
||
if not seats:
|
||
return ""
|
||
lines: list[str] = []
|
||
for sid, meta in seats:
|
||
name = str((meta or {}).get("name") or sid)
|
||
desc = str((meta or {}).get("description") or "").strip()
|
||
lines.append(f"- {sid}({name}):{desc}" if desc else f"- {sid}({name})")
|
||
roster = "\n".join(lines)
|
||
return (
|
||
"【本次圆桌可调度席位清单】派活时 agent_orchestration 的 agent_name 必须从下面"
|
||
"逐字符复制一个 id,严禁自行拼造、翻译角色名或凭记忆构造:\n"
|
||
f"{roster}"
|
||
)
|
||
|
||
|
||
async def _collect_delivery_status(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
seat_threads: dict[str, str],
|
||
*,
|
||
max_chars: int = 0,
|
||
) -> dict[str, tuple[bool, str, int, str]]:
|
||
"""读各席位 thread 的真实状态,客观判定「是否已交付 + 交付摘要」。
|
||
|
||
返回 ``{seat_id: (delivered, text, full_len, invalid_reason)}``。``text`` 是该席位**可见正文**
|
||
(按 ``max_chars`` 截断:派活轮取 600 字摘要、综合轮 ``max_chars<=0`` 取全文);``full_len`` 是
|
||
可见正文全文长度。判定依据:剥掉思考后是否有可见正文(不要求 md 文件);思考-only / 空串 /
|
||
断流都不算已交付。
|
||
|
||
走**轻量读**接口(``GET /last-ai-message?max_chars=N``):服务端只回最后一条 AI 消息(可截断),
|
||
不把整份 state 序列化经 loopback 传回 —— 派活轮每席位只取 600 字,大幅压低原先 ~11s 的
|
||
``payload_built``。任一席位读失败按「未交付」保守处理,绝不阻断。
|
||
|
||
这是「确定性双保险」的数据源:总控弱模型会臆造「三方已交付」,这里由网关**客观核对**每个席位
|
||
thread 的实际产出,再把结论喂回总控输入,总控以系统核对为准,不靠自己读历史猜。
|
||
"""
|
||
async def _one(sid: str, tid: str) -> tuple[str, tuple[bool, str, int, str]]:
|
||
try:
|
||
resp = await client.get(
|
||
f"{base}/api/threads/{tid}/last-ai-message",
|
||
params={"max_chars": max_chars},
|
||
headers=headers,
|
||
)
|
||
if resp.status_code != 200:
|
||
return sid, (False, "", 0, "empty")
|
||
data = resp.json()
|
||
content = str(data.get("content") or "")
|
||
full_len = int(data.get("length") or len(content))
|
||
reason = str(data.get("invalid_reason") or "")
|
||
return sid, (bool(data.get("delivered")), content, full_len, reason)
|
||
except Exception:
|
||
return sid, (False, "", 0, "empty")
|
||
|
||
results = await asyncio.gather(*(_one(sid, tid) for sid, tid in seat_threads.items()))
|
||
return dict(results)
|
||
|
||
|
||
# 交付状态摘要块里,每个席位「已交付摘要」截断到的字符数(让总控有足够信息做派活决策,
|
||
# 又远小于动辄几万字的全文)。
|
||
_STATUS_EXCERPT_CHARS = 600
|
||
|
||
|
||
def _build_seat_status_block(
|
||
seats: list[tuple[str, dict[str, Any]]],
|
||
delivery: dict[str, tuple[bool, str, int, str]] | dict[str, tuple[bool, str, int]],
|
||
*,
|
||
undelivered_count: int = 0,
|
||
) -> str:
|
||
"""席位清单 + **系统客观核对的真实交付状态**,注入总控本轮输入最前面。
|
||
|
||
在 ``_build_seat_roster_block``(只给 id+名称)基础上,叠加每个席位的 ✅已交付 / ⬜未交付
|
||
标记(来自 ``_collect_delivery_status`` 对席位 thread 的实读),并明确告诉总控「以这份为准、
|
||
不要臆造」「用户点名的 ⬜ 席位必须派活」。这是消除「臆造三方已交付直接收口」的确定性手段。
|
||
|
||
``undelivered_count`` > 0 时额外加一句硬约束:还有席位未交付,**不得**急于收口。
|
||
"""
|
||
if not seats:
|
||
return ""
|
||
lines: list[str] = []
|
||
for sid, meta in seats:
|
||
name = str((meta or {}).get("name") or sid)
|
||
item = delivery.get(sid, (False, "", 0, ""))
|
||
delivered, text, full_len = item[0], item[1], item[2]
|
||
reason = item[3] if len(item) > 3 else ""
|
||
if delivered:
|
||
# text 已是服务端按 max_chars 截好的摘要;full_len > len(text) 说明被截断,加省略号。
|
||
ellipsis = "…" if full_len > len(text) else ""
|
||
tail = f" —— 交付摘要:{text}{ellipsis}" if text else ""
|
||
lines.append(f"- {sid}({name}):✅ 已交付{tail}")
|
||
elif reason == "thinking_only":
|
||
lines.append(f"- {sid}({name}):⚠️ 无有效正文(仅思考或中断,不算交付)")
|
||
else:
|
||
lines.append(f"- {sid}({name}):⬜ 未交付")
|
||
roster = "\n".join(lines)
|
||
pending_rule = (
|
||
(
|
||
f"\n⚠️ 当前还有 {undelivered_count} 个未交付席位(⬜ 未交付或 ⚠️ 仅思考/中断)。"
|
||
"**信息尚未集齐,严禁现在收口/输出最终方案**,"
|
||
"也不得声称「各席位均已发言」。请**优先继续派活**给这些席位,把它们的真实可见正文拿到手再说综合。"
|
||
"思考过程、半截输出都不算交付。"
|
||
)
|
||
if undelivered_count > 0
|
||
else ""
|
||
)
|
||
return (
|
||
"【本次圆桌席位与真实交付状态(系统客观核对,以此为准,严禁臆造)】\n"
|
||
"派活时 agent_orchestration 的 agent_name 必须从下面逐字符复制一个 id;"
|
||
"只有标 ✅ 的才算已交付(可见正文,不要求 md 文件);标 ⬜ / ⚠️ 的都还没交付——\n"
|
||
"若用户想要的席位是 ⬜ 或 ⚠️,你**必须**派活给它拿到真实可见正文,**不得**跳过它直接收口或代答;"
|
||
"也**不得**声称任何 ⬜ / ⚠️ 席位已交付。下面只是**摘要**,做最终综合时以系统额外提供的"
|
||
"「完整交付」为准:\n"
|
||
f"{roster}"
|
||
f"{pending_rule}"
|
||
)
|
||
|
||
|
||
def _build_full_deliverables_block(
|
||
seats: list[tuple[str, dict[str, Any]]],
|
||
delivery: dict[str, tuple[bool, str, int, str]] | dict[str, tuple[bool, str, int]],
|
||
) -> str:
|
||
"""综合轮专用:把各已交付席位的**完整交付正文**拼成一块,供总控写最终方案。
|
||
|
||
只在前端标记的「综合轮」(synthesis_mode)注入,平时不进总控输入 —— 这样派活决策轮
|
||
的上下文保持精简、响应快,只有真要收口时才承担一次全文的体量。未交付席位不列入。
|
||
"""
|
||
sections: list[str] = []
|
||
for sid, meta in seats:
|
||
item = delivery.get(sid, (False, "", 0, ""))
|
||
delivered, full_text = item[0], item[1]
|
||
if not delivered or not full_text:
|
||
continue
|
||
name = str((meta or {}).get("name") or sid)
|
||
sections.append(f"━━ {name}({sid})的完整交付 ━━\n{full_text}")
|
||
if not sections:
|
||
return ""
|
||
body = "\n\n".join(sections)
|
||
return (
|
||
"【各席位完整交付正文(综合专用,做最终方案时以此为准)】\n"
|
||
"请基于下列各席位的真实完整交付,综合输出结构化的最终方案,不要遗漏关键内容,也不要臆造:\n"
|
||
f"{body}"
|
||
)
|
||
|
||
|
||
async def _leader_run(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
agent_threads: dict[str, str],
|
||
agent_name: str,
|
||
new_message: str,
|
||
skill_stop_names: list[str],
|
||
model_name: str,
|
||
seat_meta: dict[str, dict[str, Any]] | None = None,
|
||
synthesis_mode: bool = False,
|
||
suppress_pre_broadcast: bool = True,
|
||
files: list[dict[str, Any]] | None = None,
|
||
) -> AsyncIterator[str | dict[str, Any]]:
|
||
"""总控(leader)单轮:透传上游 SSE + 末尾 yield 一个 dict 状态帧(由路由层 ``json.dumps``)。
|
||
|
||
时序要点:
|
||
1. 总控输入只进入总控 thread,绝不广播到席位 thread;
|
||
2. 跑总控 thread 的 stream,yield 每一行原始 SSE;
|
||
3. 解析 messages → 澄清 / 空 dispatch(共识)/ 派活列表并返回。
|
||
|
||
``skill_stop_names`` 会与 ``agent_orchestration`` 合并,确保 middleware 在工具调用后停图。
|
||
|
||
``seat_meta``:``{seat_id: {name, description}}``。非空时,本函数会在跑总控前**实读各席位
|
||
thread 的真实交付状态**,把「席位清单 + ✅已交付/⬜未交付 + 摘要」拼到总控本轮输入最前面
|
||
(见 ``_collect_delivery_status`` / ``_build_seat_status_block``)。这是消除「总控臆造
|
||
三方已交付、一交付就收口」的确定性双保险:总控以系统客观核对为准,不靠自己读历史猜。
|
||
|
||
``synthesis_mode``:前端在「派活后让总控判断继续/综合」的轮次置真。为真时**额外**把各
|
||
已交付席位的**完整交付正文**注入本轮输入(``_build_full_deliverables_block``),供总控写
|
||
最终方案;平时(派活决策轮)只给摘要,让总控 thread 与每轮上下文保持精简、响应快。
|
||
|
||
``suppress_pre_broadcast`` 仅为兼容旧客户端保留,服务端始终按抑制处理。角色隔离不能由
|
||
客户端布尔值决定:总控 prompt 可能含 ``agent_orchestration``,进入席位历史会诱导弱模型
|
||
模仿总控加载编排能力。
|
||
"""
|
||
t0 = time.perf_counter()
|
||
# 实读各席位 thread 的真实交付状态,构造「席位清单 + 客观交付状态(摘要)」块;综合轮再额外
|
||
# 拼上各已交付席位的完整交付正文。失败/无 seat_meta 时退化为不注入(leader_input ==
|
||
# new_message),绝不阻断 leader run。这些只进总控**本轮 run 输入**,不广播给席位。
|
||
seat_status_block = ""
|
||
full_deliverables_block = ""
|
||
if seat_meta:
|
||
seat_ids = [name for name in agent_threads if name != agent_name]
|
||
if seat_ids:
|
||
# 第 1 阶段:**轻量**读各席位最后一条 AI 交付,只取 ≤_STATUS_EXCERPT_CHARS 字 ——
|
||
# 派活轮只需 ✅/⬜ + 摘要,绝不下载几万字全文,把 payload_built 从 ~11s 压下来。
|
||
delivery: dict[str, tuple[bool, str]] = {}
|
||
try:
|
||
delivery = await _collect_delivery_status(
|
||
client, base, headers, {sid: agent_threads[sid] for sid in seat_ids},
|
||
max_chars=_STATUS_EXCERPT_CHARS,
|
||
)
|
||
except Exception:
|
||
logger.warning("[leader-roster] 收集席位交付状态失败,退化为仅席位清单", exc_info=True)
|
||
undelivered = [sid for sid in seat_ids if not delivery.get(sid, (False, ""))[0]]
|
||
seats_meta_list = [(sid, seat_meta.get(sid, {})) for sid in seat_ids]
|
||
seat_status_block = _build_seat_status_block(seats_meta_list, delivery, undelivered_count=len(undelivered))
|
||
# 第 2 阶段(仅「综合轮 且 所有席位都已交付」):再**全文**取各已交付席位的交付,拼综合块。
|
||
# 还有 ⬜ 未交付席位时不注入 —— 否则"综合输出最终方案"的指令会怂恿弱模型在席位没集齐时
|
||
# 就提前收口(实测:3/5 交付就臆称"三方已发言"草草收口)。全文读只在这一轮才付出。
|
||
if synthesis_mode and not undelivered:
|
||
try:
|
||
full_delivery = await _collect_delivery_status(
|
||
client, base, headers,
|
||
{sid: agent_threads[sid] for sid in seat_ids if delivery.get(sid, (False, ""))[0]},
|
||
max_chars=0,
|
||
)
|
||
full_deliverables_block = _build_full_deliverables_block(seats_meta_list, full_delivery)
|
||
except Exception:
|
||
logger.warning("[leader-roster] 综合轮取全文交付失败,退化为仅摘要", exc_info=True)
|
||
logger.info(
|
||
"[leader-roster] %s 注入席位交付状态:%s synthesis_mode=%s undelivered=%d full_deliverables=%s",
|
||
agent_name,
|
||
{sid: delivery.get(sid, (False, ""))[0] for sid in seat_ids},
|
||
synthesis_mode,
|
||
len(undelivered),
|
||
bool(full_deliverables_block),
|
||
)
|
||
leader_input = "\n\n".join(
|
||
part for part in (seat_status_block, full_deliverables_block, new_message) if part
|
||
)
|
||
|
||
# 兼容旧请求字段但不再信任它。总控输入与派活说明始终只留在总控 thread;席位需要的上下文
|
||
# 由总控下发的具体 task、peer_deliveries 和席位交付广播提供。
|
||
_ = suppress_pre_broadcast
|
||
thread_id = agent_threads[agent_name]
|
||
# 角色参数取自单一数据源 roundtable_run_policy(leader:禁 web_search、停
|
||
# agent_orchestration 图、关 thinking + low reasoning——短决策、首字节快)。
|
||
# 调用方可再叠加自己的 skill_stop。后台进程内网关用同一份 policy,保证传参一致。
|
||
policy = roundtable_run_policy("leader")
|
||
effective_stop_names = merge_skill_stop_names(policy.skill_stop_names, skill_stop_names)
|
||
# 带了上传文件时,给总控解锁「读」类工具(leader policy 默认禁 read_file/ls/view_image),
|
||
# 否则文件清单注入了却无工具读取。写/派活类仍按 policy 封锁,不放开。
|
||
leader_excluded = policy.excluded_tools
|
||
if files:
|
||
_read_tools = {"read_file", "ls", "grep", "view_image"}
|
||
leader_excluded = [t for t in leader_excluded if t not in _read_tools]
|
||
|
||
# 模型自动容错:按 config.yaml models[] 顺序的候选链。当前模型出问题(兜底错误文案 /
|
||
# 既无正文又无派活)时切到下一个模型重跑(见循环内检测)。``_mk_body`` 用候选模型重建 body。
|
||
model_chain = fallback_model_chain(model_name) or [model_name]
|
||
model_idx = 0
|
||
current_model = model_chain[0]
|
||
|
||
def _mk_body(m: str) -> Any:
|
||
return _run_payload(
|
||
agent_name=agent_name,
|
||
model_name=m,
|
||
thread_id=thread_id,
|
||
# 注入了席位清单的输入(leader_input);seat_roster 为空时等同 new_message。
|
||
new_message=leader_input,
|
||
excluded_tools=leader_excluded,
|
||
skill_stop_names=effective_stop_names,
|
||
thinking_enabled=policy.thinking_enabled,
|
||
reasoning_effort=policy.reasoning_effort,
|
||
# 人工干预/恢复会重发 leader run 到同一总控 thread;抢占掉没结束的旧 run,避免 409。
|
||
multitask_strategy="rollback",
|
||
# 用户随干预上传的参考文件 → 总控可用 read_file 等工具读取后再派活。
|
||
files=files,
|
||
)
|
||
|
||
body = _mk_body(current_model)
|
||
t_payload_built = time.perf_counter()
|
||
logger.info(
|
||
"[leader-timing] %s payload_built=%.3fs",
|
||
agent_name,
|
||
t_payload_built - t0,
|
||
)
|
||
|
||
# 单例总控是弱模型(deepseek-chat),已观测到它把 SOUL 里的示例 id 当真实席位
|
||
# 照抄、或臆造同款 32 位 hex —— 这些 id 不在 agent_threads 里,直接漏给前端会让
|
||
# /run/stream 报 400 把整个 Step 2 打挂。这里把「本次真实可派席位」作为 roster
|
||
# (agent_threads 去掉总控自身),派活前就地校验:非法 id 丢弃 + 记日志;若**整轮
|
||
# 全是非法 id**,向总控 thread 回灌一条系统纠正消息(列出真实 id),再重跑一轮让它
|
||
# 自纠(最多重试 1 次,避免弱模型反复臆造导致死循环)。
|
||
valid_seat_ids: set[str] = {name for name in agent_threads if name != agent_name}
|
||
_MAX_DISPATCH_RETRIES = 1
|
||
|
||
attempt = 0
|
||
while True:
|
||
captured: list[str] = []
|
||
first_line_ts: float | None = None
|
||
line_count = 0
|
||
async for line, done in _stream_upstream(client, base, headers, thread_id, body):
|
||
if line is not None:
|
||
if first_line_ts is None:
|
||
first_line_ts = time.perf_counter()
|
||
logger.info(
|
||
"[leader-timing] %s first_upstream_line=%.3fs since_entry=%.3fs",
|
||
agent_name,
|
||
first_line_ts - t_payload_built,
|
||
first_line_ts - t0,
|
||
)
|
||
line_count += 1
|
||
yield line
|
||
elif done is not None:
|
||
captured = done
|
||
t_stream_done = time.perf_counter()
|
||
logger.info(
|
||
"[leader-timing] %s upstream_stream_complete=%.3fs lines=%d since_entry=%.3fs",
|
||
agent_name,
|
||
t_stream_done - (first_line_ts or t_payload_built),
|
||
line_count,
|
||
t_stream_done - t0,
|
||
)
|
||
|
||
messages = _parse_last_messages(captured)
|
||
if messages is None:
|
||
# 流被截断(多半被 multitask_strategy=rollback 抢占 / 客户端断开 / abort)。
|
||
# 这不是真正的上游错误——静默收尾,不向前端抛 502。该 run 对应的 SSE 在抢占
|
||
# 场景下通常已被前端 abortAllInFlight 放弃;即便没被放弃,前端拿到"无终态帧"
|
||
# 也只会当作空轮次,不会冒出错误条。
|
||
# WARNING 级:正常应是被 rollback 抢占;但若上游 run 因别的原因(如 DB 连接断、
|
||
# LLM 报错)截断流也会走这里 → 静默收尾会把它表现成前端"(无文本输出)"。留个
|
||
# 显眼日志,便于把"真异常被静默吞掉"和"正常抢占"区分开排查。
|
||
logger.warning(
|
||
"[leader-timing] %s upstream truncated (no values frame) — ending quietly "
|
||
"(expected on rollback preemption; otherwise check the upstream run's server-side error)",
|
||
agent_name,
|
||
)
|
||
return
|
||
|
||
dispatch_msg = _find_dispatch_message(messages)
|
||
leader_text = _flatten_content(dispatch_msg.get("content", ""))
|
||
tool_calls = dispatch_msg.get("tool_calls") or []
|
||
|
||
# 模型自动容错:本轮总控产出是「中间件兜底错误文案」或「既无正文又无派活」(模型没干活)
|
||
# → 若还有备选模型,切到下一个重跑(不消耗派活纠正预算)。先发 model_switch 帧让前端把
|
||
# 这条失败气泡重置,再用新模型重新流式。最后一个模型仍失败则照常往下走(错误文案当共识展示)。
|
||
fail_reason = is_llm_error_message(dispatch_msg)
|
||
if fail_reason is None and not (leader_text or "").strip() and not tool_calls:
|
||
fail_reason = "empty"
|
||
if fail_reason and model_idx + 1 < len(model_chain):
|
||
model_idx += 1
|
||
next_model = model_chain[model_idx]
|
||
logger.warning("[leader-fallback] %s 模型 %s 失败(%s),切换到 %s", agent_name, current_model, fail_reason, next_model)
|
||
yield {
|
||
"status": "model_switch",
|
||
"role": "leader",
|
||
"agent_name": agent_name,
|
||
"failed_model": current_model,
|
||
"next_model": next_model,
|
||
"reason": reason_text(fail_reason),
|
||
}
|
||
current_model = next_model
|
||
body = _mk_body(current_model)
|
||
continue
|
||
|
||
# 澄清分支:与 intent.py 相同字段;status 为字符串 "clarification"(非数组),
|
||
# 前端据此暂停 runOrchestration 循环,展示 Step 2 澄清条。
|
||
clarification_call = next(
|
||
(tc for tc in tool_calls if (tc.get("name") or "").lower() == "ask_clarification"),
|
||
None,
|
||
)
|
||
if clarification_call is not None:
|
||
cargs = clarification_call.get("args") or {}
|
||
question = str(cargs.get("question") or "").strip()
|
||
follow_up = _find_trailing_ai_text(messages, exclude_text=leader_text)
|
||
yield {
|
||
"status": "clarification",
|
||
"content": follow_up or question or leader_text,
|
||
"question": question,
|
||
"clarification_type": str(cargs.get("clarification_type") or ""),
|
||
"clarification_context": str(cargs.get("context") or ""),
|
||
"options": cargs.get("options") or [],
|
||
"allow_custom": bool(cargs.get("allow_custom", True)),
|
||
"allow_multiple": resolve_allow_multiple(cargs),
|
||
"agent_name": agent_name,
|
||
"model_used": current_model,
|
||
}
|
||
return
|
||
|
||
if not tool_calls:
|
||
# 空 status 数组 = 总控不再派活,前端视为共识(hasConsensus)。
|
||
yield {"status": [], "content": leader_text, "agent_name": agent_name, "model_used": current_model}
|
||
return
|
||
|
||
# 派活前按真实 roster 校验:合法保留、非法丢弃(只丢非法,不殃及同轮合法席位)。
|
||
valid_dispatch, invalid_names = _partition_dispatch(tool_calls, valid_seat_ids)
|
||
if invalid_names:
|
||
logger.warning(
|
||
"[leader-dispatch] %s 派活到 %d 个不存在的席位 id %s;真实席位=%s(attempt=%d)",
|
||
agent_name,
|
||
len(invalid_names),
|
||
invalid_names,
|
||
sorted(valid_seat_ids),
|
||
attempt,
|
||
)
|
||
|
||
# 整轮全是非法 id 且还有重试预算 → 回灌纠正消息 + 重跑一轮让总控自纠。
|
||
if invalid_names and not valid_dispatch and attempt < _MAX_DISPATCH_RETRIES:
|
||
try:
|
||
await _append_thread_message(
|
||
client,
|
||
base,
|
||
headers,
|
||
thread_id,
|
||
_build_dispatch_correction(invalid_names, valid_seat_ids),
|
||
)
|
||
except Exception as exc:
|
||
logger.warning("[leader-dispatch] 回灌纠正消息失败: %s", exc)
|
||
attempt += 1
|
||
continue
|
||
|
||
# 非法 id 已被丢弃;若 roster 校验后无合法派活(且重试已用尽),退化为空共识,
|
||
# 而不是把假 id 漏给前端再报 400。
|
||
dispatched: list[list[str]] = [[sub_name, task_text] for sub_name, task_text in valid_dispatch]
|
||
# 关键路径优先:yield 派活帧,让前端立刻启动席位 special run。总控派活说明不广播给
|
||
# 任何席位,避免席位从历史中模仿总控角色;席位间承上依靠完成交付广播 / peer_deliveries。
|
||
yield {"status": dispatched, "content": leader_text, "agent_name": agent_name, "model_used": current_model}
|
||
return
|
||
|
||
|
||
async def _special_run(
|
||
client: httpx.AsyncClient,
|
||
base: str,
|
||
headers: dict[str, str],
|
||
agent_threads: dict[str, str],
|
||
agent_name: str,
|
||
new_message: str,
|
||
skill_stop_names: list[str],
|
||
model_name: str,
|
||
*,
|
||
display_name: str = "",
|
||
no_file: bool = False,
|
||
peer_deliveries: list[dict[str, Any]] | None = None,
|
||
files: list[dict[str, Any]] | None = None,
|
||
thinking_enabled: bool | None = None,
|
||
reasoning_effort: str | None = None,
|
||
subagent_enabled: bool = False,
|
||
seat_skill_directive: bool = False,
|
||
business_code: str = "",
|
||
) -> AsyncIterator[str | dict[str, Any]]:
|
||
"""子智能体(special)单轮:执行总控下发的 task 文本,交付 content。
|
||
|
||
- ``excluded_tools`` 禁用派活/澄清等,避免席位抢总控活;
|
||
- 终态 ``status`` 为描述性字符串(非数组),``content`` 为交付正文;
|
||
- **后广播**在 yield 终态帧之后 await,保证其它席位 state 里有完成摘要,
|
||
同时尽量让前端先收到结果关闭 loading。
|
||
|
||
``no_file``:DAG 并行批次里的席位置真 —— 用 ``seat_parallel`` policy 额外禁
|
||
write_file / bash / str_replace,避免并行席位同时写沙箱打架(§4)。
|
||
|
||
``peer_deliveries``:前序席位的**完整交付** ``[{name, content}, ...]``。注入到
|
||
run context 的 ``roundtable_peer_deliveries``,供席位用 ``read_peer_delivery`` 工具
|
||
**按需**取全文(平时只看广播摘要,§12.9 传摘要按需传全文)。不进 prompt,不耗 token。
|
||
|
||
``thinking_enabled`` / ``reasoning_effort`` / ``subagent_enabled``:圆桌第二步「席位
|
||
执行模式」(快速/思考/专业问答/多智能体问答)派生的覆盖值。**仅 special 席位**接收:
|
||
前两者为 None 时回退到 policy 默认(向后兼容,旧前端不传 → 维持 thinking + medium);
|
||
``subagent_enabled`` 仅 ultra 档为 True,给席位解锁 ``task`` 子代理工具。**总控
|
||
(leader)不受影响**,始终用其快速派活策略。
|
||
"""
|
||
t0 = time.perf_counter()
|
||
thread_id = agent_threads[agent_name]
|
||
# 角色参数取自单一数据源 roundtable_run_policy(seat / seat_parallel / report)。
|
||
# report 在 seat 基础上额外禁 bash / str_replace;seat_parallel 额外禁
|
||
# write_file / bash / str_replace(理由见 roundtable_run_policy 模块注释)。
|
||
# 后台进程内网关用同一份 policy,保证「后台挂起」与「前端手动调用」传参一致。
|
||
if agent_name == REPORT_AGENT_ID:
|
||
role = "report"
|
||
elif agent_name == SUMMARY_AGENT_ID:
|
||
role = "summary"
|
||
elif agent_name == ACTION_PLAN_AGENT_ID:
|
||
role = "action_plan"
|
||
elif agent_name == DASHBOARD_AGENT_ID:
|
||
# 大屏多页数据智能体:纯「研讨成果 → report-json」提炼,禁用一切工具。
|
||
role = "dashboard"
|
||
elif agent_name == STRUCTURE_AGENT_ID:
|
||
# 结构化抽取智能体在「入库」轮:真正调用入库技能(read_file 读 SKILL.md +
|
||
# bash/write_file 落地),故走放开技能执行链路的 ``ingest`` policy。
|
||
role = "ingest"
|
||
elif no_file:
|
||
role = "seat_parallel"
|
||
else:
|
||
role = "seat"
|
||
policy = roundtable_run_policy(role)
|
||
# 带了上传文件时,给席位解锁「读」类工具(view_image 默认被禁;read_file 本就可用),
|
||
# 让席位能读取/识别用户上传的文件。写/派活类仍按 policy 封锁。
|
||
seat_excluded = policy.excluded_tools
|
||
if files:
|
||
_read_tools = {"read_file", "ls", "grep", "view_image"}
|
||
seat_excluded = [t for t in seat_excluded if t not in _read_tools]
|
||
# 弱模型补偿:仅对真正的研讨席位(seat / seat_parallel;report/summary/dashboard/ingest
|
||
# 等 Step3 单例不在此列)把其配置技能(名称 + SKILL.md 路径 + 必须先 read_file 再按其
|
||
# 流程取信息的硬指令)显式拼进本轮任务,避免弱模型无视系统提示里的技能块(见
|
||
# roundtable_seat_skills 模块说明)。默认关闭:是否注入 = 该席位 agent 的开关
|
||
# (config.yaml)OR 本次会商所选业务链条的开关(seat_skill_directive 透传)。无配置
|
||
# 技能 / 两开关都关时原样返回。
|
||
seat_message = new_message
|
||
if role in ("seat", "seat_parallel"):
|
||
seat_message = append_seat_skill_directive(agent_name, new_message, chain_enabled=seat_skill_directive)
|
||
# 角色边界放在所有技能强化之后,确保本轮最后一条指令明确:席位永远不能加载/调用编排能力。
|
||
seat_message = append_seat_role_boundary(seat_message)
|
||
# 深链接业务规范(仅 summary 单例):把该业务的「业务链完整性核对」指令拼进本轮任务,强制总结报告
|
||
# 核对产出相对业务链完不完整。business_code 为空(普通会商)/ 非写实业务 → directive 为 "",原样不变。
|
||
elif role == "summary" and business_code:
|
||
directive = build_summary_directive(business_code)
|
||
if directive:
|
||
seat_message = f"{new_message}\n\n---\n\n{directive}"
|
||
# 模型自动容错:按 config.yaml models[] 顺序的候选链。某模型出问题(兜底错误文案;
|
||
# 真正的研讨席位还含「交付为空」)→ 发 model_switch 帧让前端重置该席位气泡,再用下一个
|
||
# 模型重跑。Step3 单例(report/summary/dashboard/ingest)的产物是落盘文件,正文可能很短,
|
||
# 故**不**把「正文为空」当失败(与后台 run_report/summary 的 validate=None 一致),只在真正
|
||
# 的模型错误时换模型。
|
||
model_chain = fallback_model_chain(model_name) or [model_name]
|
||
empty_is_failure = role in ("seat", "seat_parallel")
|
||
|
||
def _mk_body(m: str) -> Any:
|
||
return _run_payload(
|
||
agent_name=agent_name,
|
||
model_name=m,
|
||
thread_id=thread_id,
|
||
new_message=seat_message,
|
||
excluded_tools=seat_excluded,
|
||
skill_stop_names=merge_skill_stop_names(policy.skill_stop_names, skill_stop_names),
|
||
# 席位执行模式覆盖 policy 的 thinking/reasoning(None → 回退 policy 默认);
|
||
# subagent 仅 ultra 档置真。工具裁剪(excluded_tools)仍完全由 policy 决定。
|
||
thinking_enabled=(policy.thinking_enabled if thinking_enabled is None else thinking_enabled),
|
||
reasoning_effort=(policy.reasoning_effort if reasoning_effort is None else reasoning_effort),
|
||
subagent_enabled=subagent_enabled,
|
||
# ultra 席位才会派子智能体 → 才需要 custom 流回传子任务实时进度(task_running 等)。
|
||
stream_custom=subagent_enabled,
|
||
# 干预/重新派活会把同一席位 thread 再跑一次;抢占掉没结束的旧 run,避免 409。
|
||
multitask_strategy="rollback",
|
||
# 前序席位完整交付 → run context,供 read_peer_delivery 工具按需取全文(§12.9)。
|
||
extra_context=(
|
||
{"roundtable_peer_deliveries": peer_deliveries} if peer_deliveries else None
|
||
),
|
||
# 用户随干预上传的参考文件 → 席位可用 read_file 等工具读取。
|
||
files=files,
|
||
)
|
||
|
||
t_payload_built = time.perf_counter()
|
||
logger.info(
|
||
"[special-timing] %s payload_built=%.3fs",
|
||
agent_name,
|
||
t_payload_built - t0,
|
||
)
|
||
# 优先用 display_name(中文友好,例如"情报收集")提示 leader,id 放括号备查;
|
||
# display_name 缺失时退化为仅 agent_name —— 与旧行为一致。
|
||
speaker_label = f"{display_name}({agent_name})" if display_name else agent_name
|
||
|
||
for idx, candidate in enumerate(model_chain):
|
||
body = _mk_body(candidate)
|
||
captured: list[str] = []
|
||
first_line_ts: float | None = None
|
||
line_count = 0
|
||
async for line, done in _stream_upstream(client, base, headers, thread_id, body):
|
||
if line is not None:
|
||
if first_line_ts is None:
|
||
first_line_ts = time.perf_counter()
|
||
logger.info(
|
||
"[special-timing] %s first_upstream_line=%.3fs since_entry=%.3fs",
|
||
agent_name,
|
||
first_line_ts - t_payload_built,
|
||
first_line_ts - t0,
|
||
)
|
||
line_count += 1
|
||
yield line
|
||
elif done is not None:
|
||
captured = done
|
||
t_stream_done = time.perf_counter()
|
||
logger.info(
|
||
"[special-timing] %s upstream_stream_complete=%.3fs lines=%d since_entry=%.3fs",
|
||
agent_name,
|
||
t_stream_done - (first_line_ts or t_payload_built),
|
||
line_count,
|
||
t_stream_done - t0,
|
||
)
|
||
|
||
messages = _parse_last_messages(captured)
|
||
if messages is None:
|
||
# 流被截断(网断 / rollback 抢占)。不抛 502,但席位路径要明确告诉前端:本轮不算交付,
|
||
# 避免总控把 checkpoint 里残留的思考/半截当成已完成。
|
||
logger.warning(
|
||
"[special-timing] %s upstream truncated (no values frame) — ending as invalid_delivery "
|
||
"(expected on rollback preemption; otherwise check the upstream run's server-side error)",
|
||
agent_name,
|
||
)
|
||
if empty_is_failure:
|
||
yield {
|
||
"status": "invalid_delivery",
|
||
"reason": "truncated",
|
||
"content": "",
|
||
"agent_name": agent_name,
|
||
"model_used": candidate,
|
||
}
|
||
return
|
||
# Sub-agent's final deliverable is visible body after the last human (this turn).
|
||
last_msg: dict[str, Any] = {}
|
||
for msg in reversed(messages):
|
||
if isinstance(msg, dict) and (msg.get("type") or "").lower() in ("ai", "aimessage", "aimessagechunk"):
|
||
last_msg = msg
|
||
break
|
||
content = _flatten_content(last_msg.get("content", ""))
|
||
visible = visible_seat_text(content)
|
||
if not visible:
|
||
visible = _visible_delivery_from_messages(messages)
|
||
delivery_kind = classify_seat_delivery(content) if content.strip() else ("valid" if visible else "empty")
|
||
|
||
# 模型失败检测:兜底错误文案;研讨席位再加「无可见正文」(思考不算)。还有备选模型则换。
|
||
fail_reason = is_llm_error_message(last_msg)
|
||
if fail_reason is None and empty_is_failure and delivery_kind != "valid":
|
||
fail_reason = "thinking_only" if delivery_kind == "thinking_only" else "empty"
|
||
if fail_reason and idx + 1 < len(model_chain):
|
||
next_model = model_chain[idx + 1]
|
||
logger.warning("[special-fallback] %s 模型 %s 失败(%s),切换到 %s", agent_name, candidate, fail_reason, next_model)
|
||
yield {
|
||
"status": "model_switch",
|
||
"role": role,
|
||
"agent_name": agent_name,
|
||
"failed_model": candidate,
|
||
"next_model": next_model,
|
||
"reason": reason_text(fail_reason),
|
||
}
|
||
continue
|
||
|
||
if empty_is_failure and delivery_kind != "valid":
|
||
logger.warning("[special-fallback] %s 所有模型均未产出可见正文(%s)", agent_name, delivery_kind)
|
||
yield {
|
||
"status": "invalid_delivery",
|
||
"reason": delivery_kind,
|
||
"content": "",
|
||
"agent_name": agent_name,
|
||
"model_used": candidate,
|
||
}
|
||
return
|
||
|
||
# 成功:发交付终态帧 + detached 后广播(只广播可见正文,不含思考)。
|
||
_prefix = f"子智能体{speaker_label}完成{new_message}工作,交付内容为:"
|
||
brief_msg = _truncate_broadcast_content(_prefix, visible or content)
|
||
yield {
|
||
"status": f"子智能体{speaker_label}完成{new_message}工作",
|
||
"content": visible or content,
|
||
"agent_name": agent_name,
|
||
"model_used": candidate,
|
||
}
|
||
_spawn_detached_broadcast(base, headers, brief_msg, agent_threads, [agent_name])
|
||
return
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# POST /api/multi-agent/run/stream — leader / special 单轮 SSE
|
||
# ---------------------------------------------------------------------------
|
||
|
||
|
||
@router.post("/run/stream")
|
||
async def run_multi_agent_stream(request: Request, payload: dict = Body(...)) -> StreamingResponse:
|
||
"""驱动总控或子智能体的一轮 LangGraph run。
|
||
|
||
请求体::
|
||
|
||
{
|
||
"agent_type": "leader" | "special",
|
||
"d_agent_thread_id": { "<agent_id>": "<thread_id>", ... }, // init_done.thread_ids
|
||
"agent_name": "<本次运行的 agent_id>",
|
||
"new_message": "<用户消息或派活 task 文本>",
|
||
"model": "<可选>", // 不传由 lead_agent 回退到 config.yaml models[0]
|
||
"skill_stop_names": [] // 可选,leader 一般留空,由网关注入 agent_orchestration
|
||
}
|
||
|
||
响应:上游 SSE 行透传 + 末尾一条 ``data:`` JSON(leader 为 dict,special 同上)。
|
||
客户端应使用 AbortSignal 支持「挂起协商」时取消 in-flight 请求。
|
||
"""
|
||
agent_type: str = str(payload.get("agent_type") or "leader").lower()
|
||
agent_threads: dict[str, str] = payload.get("d_agent_thread_id") or {}
|
||
agent_name: str = str(payload.get("agent_name") or "").lower()
|
||
# 人类可读名,仅 special 用 —— 后端把它拼进 post-broadcast 文本,这样 leader
|
||
# 下一轮看到的是"子智能体[中文名](id)完成..."而非"子智能体[UUID]完成...";
|
||
# LLM 对 UUID 不友好,只用 id 会导致总控认不出已完成的子智能体,无限循环派活。
|
||
# 缺失时 fallback 到 agent_name,旧前端不传也不破坏。
|
||
display_name: str = str(payload.get("display_name") or "").strip()
|
||
new_message: str = str(payload.get("new_message") or "")
|
||
skill_stop_names: list[str] = payload.get("skill_stop_names") or []
|
||
# 用户随干预消息上传的参考文件(已经过 /api/threads/{thread_id}/uploads)。
|
||
# 前端传 [{filename, size, path, status}, ...],透传给 _run_payload →
|
||
# additional_kwargs.files → UploadsMiddleware 注入清单 + 解锁 read_file 等工具。
|
||
_files_raw = payload.get("files")
|
||
files: list[dict[str, Any]] = _files_raw if isinstance(_files_raw, list) else []
|
||
# 综合轮标记(仅 leader 用):前端在「派活后让总控判断继续/综合」的轮次置真,网关据此
|
||
# 额外把各已交付席位的完整交付正文注入总控输入,供其写最终方案。缺省 False(派活决策轮
|
||
# 只给摘要,保持上下文精简)。旧前端不传 → False,行为安全退化。
|
||
synthesis_mode: bool = bool(payload.get("synthesis_mode") or False)
|
||
# 并行禁写文件标记(仅 special 用):前端 runDagOrchestration 在真并行 stage(≥2 席位)
|
||
# 置真,网关据此用 seat_parallel policy 额外禁 write_file/bash/str_replace(§4)。
|
||
# 兼容 no_file 别名;缺省 False → 普通 seat,旧前端不传也不破坏。
|
||
no_file: bool = bool(payload.get("parallel_no_file") or payload.get("no_file") or False)
|
||
# 旧客户端兼容字段。服务端已强制隔离总控 thread,因此无论请求值为何都不会执行 leader→seat
|
||
# 广播;仍解析并透传只是保持调用签名稳定。
|
||
suppress_pre_broadcast: bool = bool(payload.get("suppress_pre_broadcast", True))
|
||
# 前序席位完整交付(仅 special 用):前端 DAG stage>0 席位带上此前各阶段的全文,网关注入
|
||
# run context,供 read_peer_delivery 工具按需取全文(平时只看广播摘要,§12.9)。规整成
|
||
# [{name, content}] 且过滤空项;缺省 None。
|
||
peer_deliveries: list[dict[str, Any]] | None = None
|
||
_pd_raw = payload.get("peer_deliveries")
|
||
if isinstance(_pd_raw, list):
|
||
peer_deliveries = [
|
||
{"name": str(p.get("name") or ""), "content": str(p.get("content") or "")}
|
||
for p in _pd_raw
|
||
if isinstance(p, dict) and str(p.get("name") or "").strip() and str(p.get("content") or "").strip()
|
||
] or None
|
||
# 不写死任何默认模型;空串会让 lead_agent._resolve_model_name() 回退到
|
||
# config.yaml 的 models[0],与 /api/ai-writing 等其它路由一致。
|
||
model_name: str = str(payload.get("model") or "")
|
||
# 圆桌第二步「席位执行模式」(快速/思考/专业问答/多智能体问答)派生参数。**仅 special
|
||
# 席位消费**:覆盖 seat policy 的 thinking/reasoning。thinking_enabled/reasoning_effort
|
||
# 缺省(None/空)→ 后端按 policy 默认执行(向后兼容,旧前端不传维持 thinking + medium);
|
||
# subagent_enabled 仅 ultra 档为真,给席位解锁 task 子代理工具。leader 不消费这些字段。
|
||
_te_raw = payload.get("thinking_enabled")
|
||
seat_thinking_enabled: bool | None = bool(_te_raw) if _te_raw is not None else None
|
||
_re_raw = payload.get("reasoning_effort")
|
||
seat_reasoning_effort: str | None = _re_raw if isinstance(_re_raw, str) and _re_raw else None
|
||
seat_subagent_enabled: bool = bool(payload.get("subagent_enabled") or False)
|
||
# 业务链条级「技能识别强化」开关(仅 special 席位消费):前端读所选业务链条的同名配置
|
||
# 后逐席位透传。与该席位 agent 自身的 config.yaml 开关取 OR;缺省 False(默认关)。
|
||
seat_skill_directive: bool = bool(payload.get("seat_skill_directive") or False)
|
||
# 深链接业务码(rwfx→6BF 等)。仅 summary 单例消费:非空 → 按业务链抽取规范强制注入「完整性核对」
|
||
# 指令,让总结报告核对产出相对业务链完不完整。普通会商为空 → 不注入。
|
||
business_code: str = str(payload.get("business_code") or "")
|
||
|
||
if not agent_name or agent_name not in agent_threads:
|
||
raise HTTPException(
|
||
status_code=400,
|
||
detail=f"agent_name '{agent_name}' is missing from d_agent_thread_id",
|
||
)
|
||
_validate_agent_id(agent_name, "agent_name")
|
||
for key in agent_threads:
|
||
_validate_agent_id(key, "d_agent_thread_id key")
|
||
|
||
base = _loopback_base(request)
|
||
headers = _auth_headers(request)
|
||
|
||
# 总控派活前,取本次各席位的元数据(id+名称+描述);_leader_run 会据此**实读各席位
|
||
# thread 的真实交付状态**,把「席位清单 + ✅/⬜交付状态」拼进总控本轮输入最前面 ——
|
||
# 既让弱模型每轮都能逐字复制真实 id,又以系统客观核对的交付状态压制臆造。元数据走同进程
|
||
# 批量查;任一席位查不到/不可见就退化为仅 id 清单,绝不因此阻断 leader run。
|
||
seat_meta: dict[str, dict[str, Any]] | None = None
|
||
if agent_type == "leader":
|
||
seat_ids = [k for k in agent_threads if k != agent_name]
|
||
if seat_ids:
|
||
try:
|
||
seat_meta = await _get_agents_batch_direct(request, seat_ids)
|
||
except Exception:
|
||
logger.warning("[leader-roster] 取席位元数据失败,退化为仅 id 清单", exc_info=True)
|
||
seat_meta = {sid: {"id": sid, "name": sid, "description": ""} for sid in seat_ids}
|
||
|
||
async def stream_generator() -> AsyncIterator[str]:
|
||
# TCP loopback 到 127.0.0.1:<gateway port>。client 配置见 _make_loopback_client:
|
||
# trust_env=False 防 HTTP_PROXY 拦截,timeout=None 容纳 SSE 长连接。
|
||
async with _make_loopback_client(request) as client:
|
||
try:
|
||
if agent_type == "leader":
|
||
gen = _leader_run(
|
||
client, base, headers, agent_threads, agent_name, new_message, skill_stop_names, model_name,
|
||
seat_meta=seat_meta,
|
||
synthesis_mode=synthesis_mode,
|
||
suppress_pre_broadcast=suppress_pre_broadcast,
|
||
files=files,
|
||
)
|
||
else:
|
||
gen = _special_run(
|
||
client, base, headers, agent_threads, agent_name, new_message, skill_stop_names, model_name,
|
||
display_name=display_name,
|
||
no_file=no_file,
|
||
peer_deliveries=peer_deliveries,
|
||
files=files,
|
||
thinking_enabled=seat_thinking_enabled,
|
||
reasoning_effort=seat_reasoning_effort,
|
||
subagent_enabled=seat_subagent_enabled,
|
||
seat_skill_directive=seat_skill_directive,
|
||
business_code=business_code,
|
||
)
|
||
async for chunk in gen:
|
||
if isinstance(chunk, dict):
|
||
yield f"data: {json.dumps(chunk, ensure_ascii=False)}\n\n"
|
||
else:
|
||
yield chunk if chunk.endswith("\n") else chunk + "\n"
|
||
except HTTPException as exc:
|
||
logger.warning("multi-agent stream HTTPException: %s %s", exc.status_code, exc.detail)
|
||
await record_foreground_diag(
|
||
request, stage=_stage_for_agent(agent_type, agent_name), level="error",
|
||
event="run_http_error",
|
||
message=f"「{agent_name}」运行失败 [{exc.status_code}]:{exc.detail}",
|
||
detail={"status_code": exc.status_code, "detail": str(exc.detail), "agent_type": agent_type},
|
||
agent_id=agent_name, agent_name=display_name or agent_name,
|
||
)
|
||
err_frame = {
|
||
"status": "error",
|
||
"content": f"[{exc.status_code}] {exc.detail}",
|
||
"agent_name": agent_name,
|
||
}
|
||
yield f"data: {json.dumps(err_frame, ensure_ascii=False)}\n\n"
|
||
except Exception as exc:
|
||
logger.exception("multi-agent stream unexpected failure")
|
||
await record_foreground_diag(
|
||
request, stage=_stage_for_agent(agent_type, agent_name), level="error",
|
||
event="run_unexpected",
|
||
message=f"「{agent_name}」运行未预期异常:{exc!r}",
|
||
detail={"type": type(exc).__name__, "error": str(exc), "agent_type": agent_type},
|
||
agent_id=agent_name, agent_name=display_name or agent_name,
|
||
)
|
||
err_frame = {
|
||
"status": "error",
|
||
"content": f"unexpected: {exc!r}",
|
||
"agent_name": agent_name,
|
||
}
|
||
yield f"data: {json.dumps(err_frame, ensure_ascii=False)}\n\n"
|
||
|
||
return StreamingResponse(
|
||
stream_generator(),
|
||
media_type="text/event-stream",
|
||
headers={
|
||
"Cache-Control": "no-cache",
|
||
"X-Accel-Buffering": "no",
|
||
"Connection": "keep-alive",
|
||
},
|
||
)
|