deerflow-code/offline-backend-20260512/backend/scripts/inspect_latest_roundtable_run.py
2026-09-07 18:24:55 +08:00

115 lines
4.3 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""一次性巡检:把最新圆桌草稿里「后台作业 run」的关键字段 dump 到文件,
用于验证维度3 存储 + 报告 HTML + 席位 md 文件是否都对齐了。
用法(在 backend 目录):
PYTHONPATH=. uv run python scripts/inspect_latest_roundtable_run.py
输出写到 scripts/_roundtable_run_report.txt(UTF-8),直接打开看,避免 PowerShell GBK 乱码。
"""
from __future__ import annotations
import json
import sqlite3
from pathlib import Path
DB = Path(".deer-flow/data/deerflow.db")
OUT = Path("scripts/_roundtable_run_report.txt")
def _loads(v):
if isinstance(v, (bytes, bytearray)):
v = v.decode("utf-8", "replace")
if isinstance(v, str):
try:
return json.loads(v)
except Exception:
return v
return v
def main() -> None:
con = sqlite3.connect(str(DB))
con.row_factory = sqlite3.Row
lines: list[str] = []
# 最新一条有 step2.runs 的草稿
rows = con.execute(
"SELECT id, step2, step3 FROM roundtable_drafts ORDER BY updated_at DESC LIMIT 5"
).fetchall()
draft = None
top_step3 = None
for r in rows:
s2 = _loads(r["step2"]) or {}
if isinstance(s2, dict) and s2.get("runs"):
draft = (r["id"], s2)
top_step3 = _loads(r["step3"])
break
if not draft:
lines.append("没找到带 step2.runs 的草稿。")
OUT.write_text("\n".join(lines), encoding="utf-8")
print(f"written: {OUT}")
return
draft_id, s2 = draft
runs = s2.get("runs") or []
job_run = next((x for x in runs if str(x.get("id", "")).startswith("job-")), runs[-1])
lines.append(f"draft_id = {draft_id}")
lines.append(f"run.id = {job_run.get('id')} jobStatus = {job_run.get('jobStatus')} hasConsensus = {job_run.get('hasConsensus')}")
lines.append("")
# ── 维度3:threadIds ──
lines.append("【threadIds】(应为真实 map,不是 null)")
lines.append(json.dumps(job_run.get("threadIds"), ensure_ascii=False, indent=2))
lines.append("")
# ── 维度3:各对话的 steps / agentThreadId / time ──
lines.append("【step2RoundtableDialogues 摘要】")
for d in job_run.get("step2RoundtableDialogues") or []:
steps = d.get("steps") or []
step_keys = [s.get("key") for s in steps]
lines.append(
f"- [{d.get('sender')}] time={d.get('time')!r} thread={d.get('agentThreadId')!r} "
f"steps={step_keys} content_len={len(d.get('content') or '')}"
)
lines.append("")
# ── 席位 md 文件:seatStubMessages 里有没有 write_file ──
lines.append("【seatStubMessages】(席位写的文件 → 文件卡片数据源)")
stubs = job_run.get("seatStubMessages") or {}
if not stubs:
lines.append(" (空 —— 席位没有写文件 / 没采集到)")
for tid, msgs in stubs.items():
for m in msgs:
for tc in m.get("tool_calls") or []:
args = tc.get("args") or {}
lines.append(f" thread={tid} {tc.get('name')} path={args.get('path')!r} content_len={len(str(args.get('content') or ''))}")
lines.append("")
# ── 报告 HTML(P0)──
lines.append("【step3 报告】")
step3 = job_run.get("step3") or {}
html = step3.get("html") or ""
lines.append(f" html_len = {len(html)}")
lines.append(f" 含 <html 标签 = {'<html' in html.lower()} (True=真 HTML / False=纯文字回退=P0未解决)")
lines.append(f" summary = {(step3.get('summary') or '')[:120]!r}")
lines.append("")
# ── 顶层 draft.step3(结果绘制聊天记录靠它重建报告气泡)──
lines.append("【顶层 draft.step3】(前端 hydrateStep3 用它重建报告聊天记录气泡)")
if not top_step3:
lines.append(" ❌ 空 —— 报告聊天记录加载后会丢失(本次修复点:应非空)")
else:
th = (top_step3.get("html") or "") if isinstance(top_step3, dict) else ""
lines.append(f" ✅ 已存 html_len={len(th)} summary={(top_step3.get('summary') or '')[:80]!r}" if isinstance(top_step3, dict) else f" {top_step3!r}")
OUT.write_text("\n".join(lines), encoding="utf-8")
print(f"written: {OUT.resolve()}")
if __name__ == "__main__":
main()