deerflow-code/offline-backend-20260512/config.yaml
2026-09-07 18:24:55 +08:00

445 lines
22 KiB
YAML
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

config_version: 8
log_level: info
models:
# - name: deepseek-chat
# display_name: DeepSeek Chat
# use: langchain_openai:ChatOpenAI
# model: deepseek-chat
# api_key: sk-9808fb2375534112ac275541aec14b9b
# base_url: https://api.deepseek.com
# request_timeout: 600.0
# max_retries: 2
# max_tokens: 4096
# temperature: 1.0
# supports_vision: false
# supports_thinking: true
# - name: zai-org/GLM-5-FP8
# display_name: zai-org/GLM-5-FP8
# use: langchain_openai:ChatOpenAI
# model: zai-org/GLM-5-FP8
# api_key: modalresearch_gWu46FDpVLYZWPVEA3b_2fq8Jw2UJiKMgEm_Rxmn1SA
# base_url: https://api.us-west-2.modal.direct/v1
# request_timeout: 600.0
# max_retries: 2
# max_tokens: 4096
# temperature: 1.0
# supports_vision: true
# supports_thinking: true
# - name: deepseek-v4-pro
# display_name: DeepSeek V4 Pro
# use: langchain_openai:ChatOpenAI
# model: deepseek-v4-pro
# api_key: sk-e1c22b7931dc446f8b67e57878edd105
# base_url: https://api.deepseek.com
# request_timeout: 600.0
# max_retries: 2
# max_tokens: 4096
# temperature: 1.0
# supports_vision: false
# supports_thinking: true
# - name: minimax-m2.7
# display_name: MiniMax M2.7
# use: langchain_openai:ChatOpenAI
# model: MiniMax-M2.7
# api_key: sk-cp-mrsAGtzDpUl0TpOBPVoiRG_pWF1kDDN9_mDAnONBs6HIPcNUexSKtSbx2QOxfKBvJT6Xw6hDTV6hdkJYFWfNA8UtHQ08TiGJiosdRoW6q1sGb6Cb7D85NHk
# base_url: https://api.minimaxi.com
# max_tokens: 4096
# temperature: 1.0
# supports_vision: true
# supports_thinking: true
# - name: minimax-m2.5
# display_name: MiniMax M2.5
# use: langchain_openai:ChatOpenAI
# model: MiniMax-M2.5
# api_key: sk-cp-mrsAGtzDpUl0TpOBPVoiRG_pWF1kDDN9_mDAnONBs6HIPcNUexSKtSbx2QOxfKBvJT6Xw6hDTV6hdkJYFWfNA8UtHQ08TiGJiosdRoW6q1sGb6Cb7D85NHk
# base_url: https://api.minimaxi.com
# max_tokens: 4096
# temperature: 1.0
# supports_vision: true
# supports_thinking: true
# - name: deepseek-ai/DeepSeek-OCR
# display_name: deepseek-ai/DeepSeek-OCR
# use: langchain_openai:ChatOpenAI
# model: deepseek-ai/DeepSeek-OCR
# api_key: sk-beshfxfxfrllsczpqfrgjbugfbpzxgqgedbehceytpcvvtyb
# base_url: https://api.siliconflow.cn/v1/
# max_tokens: 4096
# temperature: 1.0
# supports_vision: true
# supports_thinking: true
- name: deepseek-ai/DeepSeek-V4-Flash
display_name: deepseek-ai/DeepSeek-V4-Flash
use: langchain_openai:ChatOpenAI
model: deepseek-ai/DeepSeek-V4-Flash
api_key: sk-ighylafootabtuthwueirmoycccyqtdgbifyhozduvolkdyt
base_url: https://api.siliconflow.cn/v1/
max_tokens: 4096
temperature: 1.0
supports_vision: true
supports_thinking: true
# SiliconFlow 混合推理模型关思考用「请求体顶层 enable_thinking: false」,
# 不认 chat_template_kwargs / thinking.type 那两种嵌套形状。声明这个后,
# 网页端「关闭思考」对本模型才生效(开放接口另由 force_disable_thinking 兜底)。
when_thinking_disabled:
extra_body:
enable_thinking: false
- name: Pro/MiniMaxAI/MiniMax-M2.5
display_name: Pro/MiniMaxAI/MiniMax-M2.5
use: langchain_openai:ChatOpenAI
model: Pro/MiniMaxAI/MiniMax-M2.5
api_key: sk-ighylafootabtuthwueirmoycccyqtdgbifyhozduvolkdyt
base_url: https://api.siliconflow.cn/v1/
max_tokens: 4096
temperature: 1.0
supports_vision: true
supports_thinking: true
- name: inclusionAI/Ling-flash-2.0
display_name: inclusionAI/Ling-flash-2.0
use: langchain_openai:ChatOpenAI
model: inclusionAI/Ling-flash-2.0
api_key: sk-ighylafootabtuthwueirmoycccyqtdgbifyhozduvolkdyt
base_url: https://api.siliconflow.cn/v1/
max_tokens: 4096
temperature: 1.0
supports_vision: true
supports_thinking: true
# 识图(图片识别)配置。
# - model_name:指定一个 models[] 中的模型名,作为“默认识图大模型”。view_image 工具会把
# 图片交给该模型识别,并把识别出的文字结果返回给主对话模型 —— 这样即使主对话模型不支持
# 视觉,也能“看图”。请改成你实际用于识图的模型名。
# - 若留空 / 注释掉 model_name,则回退到主对话模型自身的视觉能力(需该模型 supports_vision: true)。
# - prompt:不带具体问题时使用的默认识图指令(可选)。
# - max_tokens:识图模型输出上限(可选,默认用该模型自身的配置)。
vision:
model_name: deepseek-ai/DeepSeek-OCR
# prompt: 请详细描述这张图片的内容。
# max_tokens: 2048
knowledge_ingest:
url: "https://ch1.b.uat.4.cn/knowes/api/knowledge-outer/add"
# 上传文件识别配置。
# - auto_convert_documents: true 时,上传的 PDF / Word / Excel / PPT 会在服务端自动转换为
# Markdown(.md),agent 通过 read_file / grep 读取其内容 —— 即“识别”文档内容。
# 纯文本类(txt / csv / md / json / 代码等)无需转换,agent 用 read_file 直接读取。
# 图片(png/jpg/webp)由 view_image + 上面的 vision 识图模型处理。
# - pdf_converter: auto(优先 pymupdf4llm,缺失时回退 markitdown)| pymupdf4llm | markitdown
uploads:
auto_convert_documents: true
pdf_converter: auto
tool_groups:
- name: web
- name: file:read
- name: file:write
- name: bash
tools:
- name: web_search
group: web
use: deerflow.community.ddg_search.tools:web_search_tool
max_results: 5
- name: ls
group: file:read
use: deerflow.sandbox.tools:ls_tool
- name: read_file
group: file:read
use: deerflow.sandbox.tools:read_file_tool
- name: glob
group: file:read
use: deerflow.sandbox.tools:glob_tool
max_results: 200
- name: grep
group: file:read
use: deerflow.sandbox.tools:grep_tool
max_results: 100
- name: write_file
group: file:write
use: deerflow.sandbox.tools:write_file_tool
- name: str_replace
group: file:write
use: deerflow.sandbox.tools:str_replace_tool
- name: bash
group: bash
use: deerflow.sandbox.tools:bash_tool
sandbox:
use: deerflow.sandbox.local:LocalSandboxProvider
allow_host_bash: true
title:
enabled: true
max_words: 6
max_chars: 60
prompt_template: |
请为下面这段对话生成一个简洁的标题,概括对话的核心主题。
要求:
- 标题必须使用简体中文
- 尽量精炼,不超过 {max_words} 个词
- 只返回标题本身,不要加引号、不要解释、结尾不要标点
User: {user_msg}
Assistant: {assistant_msg}
# 上下文压缩(对话摘要):长对话接近上下文上限时,把较早的消息总结成一段摘要,
# 只保留最近若干条原始消息,避免撑爆模型上下文窗口。
# - trigger:触发条件,可配多个,任意一个先满足就压缩(OR 关系)。
# type=tokens 按 token 总数(最准,直接防溢出)
# type=messages 按消息条数(简单直观,作兜底保险)
# type=fraction 按模型上下文窗口的比例(如 0.8);自定义 base_url 模型可能探不到
# 准确窗口,故此处用写死的 token 阈值更稳妥。
# - keep:压缩后保留多少最近的上下文(同样支持 tokens / messages / fraction)。
# - trim_tokens_to_summarize:喂给摘要模型的消息最多裁到多少 token。
# - model_name:生成摘要用的模型(留空则用主对话模型)。
skills:
path: /app/skills
security_scan:
enabled: false
# ES 灵活查询技能路由提示:开启后注入整体 system prompt;修改 prompt 可调整
# 触发边界和调用要求。关闭后该提示完全不注入。
es_query_routing:
enabled: true
prompt: |
当用户的问题需要基于 Elasticsearch 查询语句(Query DSL)灵活构造、调整或执行查询时,必须优先使用 `es_query` 技能。
典型场景包括:根据用户条件动态组合 bool/must/filter/should、范围、聚合、排序或分页等 ES 查询;或者用户直接提供、要求修改 ES 语句并据此查询数据。
执行前先读取 `es_query` 技能的 SKILL.md,严格按照技能中的流程和参数调用;不要用通用搜索替代,也不要凭记忆臆测查询结果。
仅当问题确实需要 ES 数据查询时使用;纯概念解释、普通代码说明或与 ES 无关的问题不要调用。
agents_api:
enabled: true
# 登录鉴权扩展:在基础登录之上叠加两种自定义模式。
auth_login:
# 用户名直登口令门:开关打开后,/login/<用户名> 直登必须再带 ?password=口令,
# 口令不对直接 401。password 留空则默认口令为 123ewq。仅作用于用户名直登。
username_login:
require_password: false
password: ""
# token 换登录:前端用 ?authToken=上游token 进入,后端调用 token_info_url
# (把该 token 直接拼成 Authorization: Bearer <token> 头)换出用户名,再登录;
# 用户名不在库里时自动注册。
token_login:
enabled: true
token_info_url: "https://ch1.b.uat.4.cn/consumer/login/getTokenInfo"
# 已弃用:鉴权头现由前端传入的 token 拼成 "Bearer <token>",此项不再生效,保留仅兼容。
service_authorization: "Bearer admin"
timeout_seconds: 10
# 上游是 HTTPS。UAT/内网常用自签或私有 CA 证书,默认 verify=true 会握手失败导致
# 永远 401。此处 uat 域名按自签处理,先关校验;生产请置 true 或配 ca_cert_path。
verify_ssl: false
ca_cert_path: "" # 可选:PEM 格式 CA 包路径,设置后优先生效并启用校验
# 离线测试映射:token -> 用户名。命中的 token 跳过 token_info_url 调用,直接当成
# 该用户登录——外网调不通 getTokenInfo 时用来本地全链路联调。生产留空即可(不生效)。
# 例:用 /login/任意?authToken=123ewq 进入,即以 admin 身份登录;token1=123ewq、
# username=admin 会被存下,供任务工作区跳转按钮的「追加登录态」拼到外链上。
mock_tokens:
"123ewq": "admin"
# 日志系统入口(仅管理员可见):右下角「设置」菜单中会多出一个「日志系统」入口,
# 点击后在新标签页打开下面的 url。enabled 关掉则不显示该入口。
database:
backend: mysql
mysql_url: $MYSQL_DATABASE_URL
pool_size: 10
echo_sql: false
checkpointer:
# SQLite 文件路径用相对路径,由 resolve_sqlite_conn_str 解析成绝对路径。
# Docker 部署:把容器内 .deer-flow 目录挂卷出来即可持久化(撑住容器重启)。
# 这里同时被「主 agent」和「AI 写作 graph」共用 —— LangGraph 用
# (thread_id, checkpoint_ns) 做隔离,session_id 是 uuid,不会撞。
type: postgres
# true: 新 checkpoint 写入 Postgres;false: 保留 PG 配置但继续使用 SQLite。
# 也可临时用环境变量 DEER_FLOW_USE_PG_CHECKPOINTER=0 覆盖。
use_postgres: false
connection_string: postgresql://postgres:postgres123@47.88.25.99:8432/deerflow
sqlite_connection_string: backend/.deer-flow/data/deerflow.db
dual_read_enabled: true
dual_read_sqlite_paths:
- backend/.deer-flow/data/deerflow.db
- data/.deer-flow/data/checkpoints.db
# AI 写作子系统的会话清理配置。
# 业务表(ai_writing_sessions @ MySQL) + checkpointer(checkpoints.db @ SQLite) 联动清理。
# 留意:清理走 cron 表达式调度,默认每天凌晨 3 点跑一次,按 updated_at 一刷拉清。
ai_writing:
cleanup:
enabled: true
retention_days: 7 # 保留 7 天以内的会话,更早的整条删
cron: "0 3 * * *" # 每天 03:00 本地时区
delete_only_finished: false # false=不看状态一刷拉;true=只清 done/error
batch_size: 500 # 单批最多删多少行,避免长事务
# 记忆子系统:跨会话持久化用户画像与 agent 私有记忆。
# 三层覆盖优先级(高->低):
# users/{user_id}/memory_config.json > .deer-flow/memory_user_overrides.json > 此处
memory:
enabled: true
version: v1
provider: hindsight # 'builtin' | 'hindsight'(外部 provider 名)
# builtin 始终启用,这里只决定 external 选哪个
builtin:
enabled: true
memory_char_limit: 2200
user_char_limit: 1375
deduplicate_on_load: true
hindsight: # provider=hindsight 时生效,需另行部署 Hindsight 服务
mode: local_external # 'cloud' | 'local_embedded' | 'local_external'
api_url: http://47.88.25.99:18888
api_key: "" # 部署 Hindsight 后填,或改成 $HINDSIGHT_API_KEY 并设同名环境变量
# 注意:写成 $VAR 但环境变量未设置会导致配置加载失败
bank_id_template: "deerflow-user-{user_id}"
work_bank_id_template: "deerflow-work-{user_id}-{agent_id}"
memory_mode: context # 'context' | 'tools' | 'hybrid'
bank_mission: ""
bank_retain_mission: ""
recall_budget: mid # 'low' | 'mid' | 'high'
recall_max_tokens: 4096
recall_max_input_chars: 800
auto_recall: true
auto_retain: true
retain_async: true
retain_every_n_turns: 1
retain_user_prefix: "User"
retain_assistant_prefix: "Assistant"
# background_extraction:预留功能,代码尚未实现 —— 仅设计文档存在。
# 待对应版本支持后,取消下面整段注释即可启用(漏记兜底)。
# background_extraction:
# enabled: false # 默认关闭,作为漏记兜底
# debounce_seconds: 30
# model_name: ~ # null = 用默认模型
# fact_confidence_threshold: 0.7
# max_facts_per_extraction: 10
# correction_detection: true
# reinforcement_detection: true
injection:
enabled: true
max_tokens: 2000
include_builtin: true
include_hindsight: true
context_tag: "memory-context" # 包裹 Hindsight recall 的标签名
security:
scan_content: true # 注入/外渗模式扫描
block_invisible_unicode: true
# streaming_scrubber:预留字段,代码尚未实现 —— 待版本支持后取消注释。
# streaming_scrubber: true # SSE 流前清洗 <memory-context>
log_system:
enabled: true
url: "https://your-log-system.example.com"
label: "日志系统"
# ───────────────────────────────────────────────────────────────────────────
# 操作日志(Y-Log)上报:前端直连外部接口记录用户的关键操作(审计/统计)。
# 仅在「token 登录」(存在 token1)时上报;前端经公开只读接口
# GET /api/public/config/y-log 读取本段,改这里即时生效(Gateway mtime 热重载)。
y_log:
enabled: true
write_url: https://ch1.b.uat.4.cn/api/y-log/write
# 舆情分析虚拟智能体:前端改 dist/runtime-config.js 的地址/账号/密码,
# 随请求传给 Gateway 转发;这里只配上游超时。
sentiment_agent:
timeout_seconds: 300
summarization:
# 开关已移到聊天框下方(参考文献下面)的「上下文压缩」按钮,默认关闭。
# 这里 enabled 仅作为非 Web 路径(渠道/定时任务)的后备开关;下面的 trigger/keep
# 等仅是调参,Web 端是否压缩由前端按钮(summarization_enabled)决定。
enabled: false
# model_name: deepseek-chat
trigger:
- type: tokens
value: 120000
- type: messages
value: 160
# Keep a small *token* budget of recent context (same unit as the token
# trigger, well below it). This is what the model receives after compaction:
# [summary, ~40000 tokens of recent turns]. Small keep = small model context
# (fast, no overflow) AND infrequent re-compaction — the preserved window
# grows from ~40000 up to the 120000 trigger before folding into a new summary,
# which spans many turns. The full conversation still stays in state for the
# page; only what is sent to the model is compressed.
keep:
type: tokens
value: 40000
trim_tokens_to_summarize: 8000
# ───────────────────────────────────────────────────────────────────────────
# 任务深链(goPath=rwfx):目的树闸门 + 问答后跳转。前端经公开只读接口
# GET /api/public/config/task-deeplink 读取。地址都是完整 URL(调用时仅追加
# ?taskId= / ?id=),随时改这里即可(Gateway mtime 热重载,前端无需重建)。
task_deeplink:
# 线上接口未通时用假数据;三个开关互相独立。
test: true # 目的树查询走 mock.purpose_detail 假数据;同时 rwfx 进入「3q 是否完成」预检走 mock.sq_report 假数据
delete_test: true # 删除接口走 mock.delete 假数据(该接口尚未开发完)
action_test: true # 行动→任务解析走 mock.action_detail 假数据(xdfx 行动id→任务id)
# 开放接口 POST /api/open/3qfx/ask 按 taskId 拉取任务详情拼开场白;上游不可达时打开下面开关走 mock.cop_task_detail。
cop_task_detail_test: true
cop_task_detail_url: "https://ch1.b.uat.4.cn/consumer/taskAnalyseSearch/cop-task-detail"
purpose_detail_url: "https://ch1.b.uat.4.cn/consumer/taskAnalyseSearch/getTaskPurposeDetail"
# rwfx 与 xdfx 共用同一删除接口:rwfx 追加 ?taskId=(不传 actionId);xdfx 追加 ?actionId=(不传 taskId,直接不出现)。
delete_url: "https://ch1.b.uat.4.cn/bwjt/dropSixOrSevenData"
treemap_edit_url: "https://ch1.b.uat.4.cn/web/#/task/item-aicoh/step-mode-treemap-full-edit"
treemap_next_url: "https://ch1.b.uat.4.cn/web/#/task/item-aicoh/step-mode-treemap-full-next"
# xdfx 深链传入的是「行动id」,先调该接口(追加 /{actionId})拿 data.taskId 解析出真正「任务id」。
action_detail_url: "https://ch1.b.uat.4.cn/consumer/plan/action"
# xdfx 删除前「行动历史是否存在」预检接口(追加 ?actionId=);地址待提供,留空则前端视作无历史、不阻断删除。
action_history_url: ""
# rwfx 目的树面板「3q详情」链接的目标页(地址待定,留空则前端提示未配置);跳转追加 ?id=任务id。
detail_3q_url: ""
mock:
# test=true 时原样返回给前端(结构同真实 getTaskPurposeDetail 返回)。
purpose_detail:
state: "200"
msg: "操作成功!"
data:
- purposeDto: { id: 5837, zzPurpose: "目的01", selectedFlag: 1, taskId: 3324, recDetailId: 4271 }
childrenList:
- actorDto: { id: 9101, taskId: 3324, purposeId: 5837, actorName: "行为体01", actorOrgan: "行为体01", recReason: "行为体01理由", selectedFlag: 1 }
childrenList:
- expectedBehaviorDto: { id: 6910, taskId: 3324, purposeId: 5837, behavior: "预期01", recReason: "预期01理由", selectedFlag: "1" }
childrenList:
- drivingFactorDto: { id: 7964, taskId: 3324, purposeId: 5837, drivingFactor: "预期01因素", selectedFlag: 1 }
childrenList:
- narrationDto: { id: 6500, purposeId: 5837, taskId: 3324, narration: "叙事01", recReason: "叙事01理由", selectedFlag: 1 }
childrenList: []
- actorDto: { id: 9102, taskId: 3324, purposeId: 5837, actorName: "行为体02", actorOrgan: "行为体02", recReason: "行为体02理由", selectedFlag: 1 }
childrenList:
- expectedBehaviorDto: { id: 6911, taskId: 3324, purposeId: 5837, behavior: "预期02", recReason: "预期02理由", selectedFlag: "1" }
childrenList:
- drivingFactorDto: { id: 7965, taskId: 3324, purposeId: 5837, drivingFactor: "预期02因素", selectedFlag: 1 }
childrenList:
- narrationDto: { id: 6501, purposeId: 5837, taskId: 3324, narration: "叙事02", recReason: "叙事02理由", selectedFlag: 1 }
childrenList: []
# delete_test=true 时原样返回给前端。改 success 测「成功覆盖 / 不可覆盖」两条分支。
delete:
message: "行动已经开始策划,不可覆盖"
success: true
# action_test=true 时原样返回给前端(结构同真实 /consumer/plan/action/{actionId} 返回)。
# 前端取 data.taskId 作为真正的「任务id」(xdfx 深链 ?taskId= 传入的是 data.id 这个行动id)。
action_detail:
state: "200"
msg: "操作成功!"
data:
id: 27354
purposeId: 2850
taskId: 2055
actionName: ""
actionStatus: 25
narration: ""
# test=true 时 rwfx 进入「3q 是否完成」预检(selectSqReport,外部不可用)走这份假数据。
# 结构同真实 GET /consumer/cop/selectSqReport?taskId= 返回;前端判定 data 为空 / 每条
# contentJson 都是 null/空 = 「3q 未完成」→ 弹框引导去 3qfx;否则视为已完成(正常,不弹框)。
# 默认给一条有内容的数据 = 已完成(不打扰)。要测「3q 未完成」弹框:把下面 data 改成 []。
sq_report:
state: "200"
msg: "操作成功!"
data:
- id: "sq-mock-1"
taskId: 3338
categoryType: "敌情关键事件 JSON"
contentJson: "[{\"topic\":\"示例研判事件\",\"count\":1}]"
# cop_task_detail_test=true 时原样返回给开放接口 /api/open/3qfx/ask(结构同真实 cop-task-detail 返回)。
cop_task_detail:
data:
id: 2055
overview: "暑期文旅消费高峰期间的舆情态势综合研判与引导任务"
taskDirection: "正向引导 + 风险预警"
taskName: "任务01:暑期文旅舆情综合研判"
taskContent: "围绕暑期文旅消费高峰,对重点城市、重点景区的公众情绪与舆情态势做综合研判,识别情绪拐点与潜在风险,提出正向引导与处置建议。"