220 lines
9.8 KiB
Python
220 lines
9.8 KiB
Python
"""RC-BE-018: 20 annotated tasks, deterministic scoring, blind compare, reliability."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from collections import Counter
|
|
|
|
import pytest
|
|
|
|
from app.report_collaboration.eval import (
|
|
METRIC_GATES,
|
|
RECOMMENDED_RUNTIME_PROFILE,
|
|
compare_blind,
|
|
evaluate_catalog,
|
|
evaluate_task,
|
|
get_eval_task,
|
|
load_eval_tasks,
|
|
materialize_failure,
|
|
materialize_gold,
|
|
score_report,
|
|
summarize_reliability,
|
|
)
|
|
from app.report_collaboration.eval.schema import ReportCandidate
|
|
from app.report_collaboration.eval.catalog import STANDARD_DIRECTED_MODS
|
|
from app.report_collaboration.execution.ledger import TaskResult
|
|
from app.report_collaboration.execution.member_runner import WaveExecutor
|
|
from app.report_collaboration.execution.quality_roles import QualityRoleKernel
|
|
from app.report_collaboration.interventions.intent_router import RuntimeIntentContext, RuntimeIntentRouterAgent
|
|
from deerflow.config.report_collaboration_config import ReportCollaborationConfig
|
|
from deerflow.persistence.report_collaboration import MemoryReportCollaborationStore
|
|
|
|
|
|
def test_catalog_has_twenty_balanced_tasks() -> None:
|
|
tasks = load_eval_tasks()
|
|
assert len(tasks) == 20
|
|
assert len({item.id for item in tasks}) == 20
|
|
counts = Counter(item.category for item in tasks)
|
|
assert counts == {"policy": 4, "market": 4, "enterprise": 4, "event": 4, "comparison": 4}
|
|
assert get_eval_task("ev-05").title == "新能源汽车主要品牌市场趋势"
|
|
for task in tasks:
|
|
assert task.required_angles
|
|
assert task.key_facts
|
|
assert task.forbidden_fabrications
|
|
assert task.template.sections
|
|
assert task.review_rules
|
|
assert task.acceptable_sources
|
|
assert [item.id for item in task.directed_modifications] == [item.id for item in STANDARD_DIRECTED_MODS]
|
|
|
|
|
|
def test_recommended_profile_matches_disabled_production_defaults() -> None:
|
|
config = ReportCollaborationConfig()
|
|
for key, value in RECOMMENDED_RUNTIME_PROFILE.items():
|
|
assert getattr(config, key) == value
|
|
assert config.worker_enabled is False
|
|
assert METRIC_GATES["citation_coverage"] >= 0.85
|
|
assert METRIC_GATES["angle_coverage"] == 1.0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_all_gold_reports_clear_eval_gates() -> None:
|
|
summary = await evaluate_catalog()
|
|
assert summary.tasks == 20
|
|
assert summary.passed == 20
|
|
assert summary.composite >= 0.95
|
|
assert set(summary.by_category) == {"policy", "market", "enterprise", "event", "comparison"}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_known_failures_are_reproducible() -> None:
|
|
task = get_eval_task("ev-05")
|
|
gold = await evaluate_task(task, materialize_gold(task))
|
|
assert gold.passed
|
|
fabricated = await evaluate_task(task, materialize_failure(task, "fabricated"))
|
|
assert not fabricated.passed
|
|
assert not fabricated.metric("factual_accuracy").passed
|
|
missing = await evaluate_task(task, materialize_failure(task, "missing_angle"))
|
|
assert not missing.metric("angle_coverage").passed
|
|
uncited = await evaluate_task(task, materialize_failure(task, "uncited"))
|
|
assert not uncited.metric("citation_coverage").passed
|
|
empty = score_report(task, materialize_failure(task, "empty"), directed_hits=[True, True, True])
|
|
assert not empty.passed
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_blind_compare_keeps_roundtable_failure_instance() -> None:
|
|
task = get_eval_task("ev-05")
|
|
hits = [True, True, True]
|
|
result = compare_blind(task, materialize_gold(task), materialize_failure(task, "roundtable_essay"), directed_hits=hits)
|
|
assert result.winner == "collaboration"
|
|
assert result.deltas["citation_coverage"] > 0
|
|
assert result.deltas["template_adherence"] > 0
|
|
assert not result.roundtable.passed
|
|
assert "不得仅以" in result.notes
|
|
|
|
|
|
def test_reliability_counts_twenty_fixed_runs() -> None:
|
|
task = get_eval_task("ev-05")
|
|
golds = [materialize_gold(task) for _ in range(17)]
|
|
failures = [
|
|
materialize_failure(task, "empty"),
|
|
materialize_failure(task, "illegal_json"),
|
|
materialize_gold(task).model_copy(update={"status": "failed", "notes": "early_stop"}),
|
|
]
|
|
counts = summarize_reliability(golds + failures)
|
|
assert counts.runs == 20
|
|
assert counts.completed == 17
|
|
assert counts.empty_output == 1
|
|
assert counts.illegal_json == 1
|
|
assert counts.early_stop == 1
|
|
assert counts.failure_rate == pytest.approx(3 / 20)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_rewrite_without_research_is_not_classified_as_research_again() -> None:
|
|
task = get_eval_task("ev-05")
|
|
router = RuntimeIntentRouterAgent()
|
|
rewrite = await router.route(
|
|
RuntimeIntentContext(
|
|
text="重写第二章,结论更谨慎,不要重新检索",
|
|
requirement={"topic": task.title, "required_angles": task.required_angles, "revision": 1},
|
|
plan_nodes=[{"id": "write", "role_key": "writer"}, {"id": "research", "role_key": "researcher"}],
|
|
)
|
|
)
|
|
assert rewrite.intent == "rewrite_section"
|
|
research = await router.route(
|
|
RuntimeIntentContext(
|
|
text="海外市场资料太旧,重新检索 2026 年最新信息",
|
|
requirement={"topic": task.title, "required_angles": task.required_angles, "revision": 1},
|
|
plan_nodes=[{"id": "research", "role_key": "researcher"}],
|
|
)
|
|
)
|
|
assert research.intent == "research_again"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_scripted_quality_run_covers_ev01_annotations() -> None:
|
|
task = get_eval_task("ev-01")
|
|
store = MemoryReportCollaborationStore()
|
|
session = await store.create_session(owner_id="u-eval", title=task.title, idempotency_key="eval-s")
|
|
await store.save_requirement_state(
|
|
session["id"],
|
|
requirement_json={"snapshot": {"topic": task.title, "required_angles": task.required_angles, "time_range": task.time_range, "revision": 1}},
|
|
requirement_revision=1,
|
|
status="proposal_ready",
|
|
)
|
|
plan = {
|
|
"id": "plan_eval_pipeline",
|
|
"proposal_group_id": "grp_eval",
|
|
"title": task.title,
|
|
"strategy": "balanced_review",
|
|
"summary": "eval",
|
|
"rationale": "RC-BE-018",
|
|
"recommended": True,
|
|
"estimated_duration_seconds": 300,
|
|
"estimated_cost_level": "low",
|
|
"requirement_revision": 1,
|
|
"roles": [],
|
|
"nodes": [
|
|
{"id": "research", "label": "检索", "role_key": "researcher", "output_artifact_type": "EvidenceBundle", "depends_on": [], "max_attempts": 3, "angle": task.required_angles[0]},
|
|
{"id": "verify", "label": "核验", "role_key": "verifier", "output_artifact_type": "VerificationResult", "depends_on": ["research"], "max_attempts": 3},
|
|
{"id": "analyze", "label": "分析", "role_key": "analyst", "output_artifact_type": "AngleAnalysis", "depends_on": ["verify"], "max_attempts": 3},
|
|
{"id": "write", "label": "写作", "role_key": "writer", "output_artifact_type": "ReportSectionDraft", "depends_on": ["analyze"], "max_attempts": 3},
|
|
{"id": "review", "label": "审稿", "role_key": "reviewer", "output_artifact_type": "ReviewDecision", "depends_on": ["write"], "max_attempts": 3},
|
|
],
|
|
"edges": [
|
|
{"source": "research", "target": "verify"},
|
|
{"source": "verify", "target": "analyze"},
|
|
{"source": "analyze", "target": "write"},
|
|
{"source": "write", "target": "review"},
|
|
],
|
|
"quality_gates": [],
|
|
"validation": {"ok": True, "errors": []},
|
|
"revision": 1,
|
|
"status": "proposed",
|
|
}
|
|
inserted = await store.insert_plan(session["id"], plan)
|
|
await store.select_plan(session["id"], inserted["id"], idempotency_key="eval-sel", expected_revision=None)
|
|
run = await store.create_run(session["id"], plan_id=inserted["id"], idempotency_key="eval-run", expected_revision=None)
|
|
claims = []
|
|
sources = []
|
|
angle_blob = " ".join(task.required_angles)
|
|
for fact in task.key_facts:
|
|
token = fact.acceptable_source_tokens[0]
|
|
claims.append(
|
|
{
|
|
"claim_id": fact.id,
|
|
"text": f"{fact.text} {angle_blob}",
|
|
"kind": "fact",
|
|
"source_ids": [f"src-{fact.id}"],
|
|
"excerpt": fact.text,
|
|
"published_at": "2025-06-01",
|
|
"unit": "万元" if "万" in fact.number else None,
|
|
"scope": fact.id,
|
|
}
|
|
)
|
|
sources.append({"source_id": f"src-{fact.id}", "title": f"{token} 公开口径 {angle_blob}", "excerpt": fact.text, "published_at": "2025-06-01"})
|
|
kernel = QualityRoleKernel(store)
|
|
kernel.enqueue(
|
|
"research",
|
|
TaskResult(
|
|
kind="artifact_draft",
|
|
artifact_type="EvidenceBundle",
|
|
artifact_draft={"schema_version": 1, "claims": claims, "sources": sources, "conflicts": [], "coverage_gaps": []},
|
|
),
|
|
)
|
|
status = await WaveExecutor(store, kernel).run_until_idle(run["id"])
|
|
assert status == "completed"
|
|
drafts = [item for item in await store.list_artifacts_full(run["id"]) if item["artifact_type"] == "ReportSectionDraft" and item["validation_status"] == "validated"]
|
|
assert drafts
|
|
markdown = str(drafts[-1]["content"]["markdown"])
|
|
candidate = ReportCandidate(
|
|
source="collaboration",
|
|
markdown=markdown,
|
|
source_index=[{"id": fact.id, "title": f"{fact.acceptable_source_tokens[0]} 公开口径", "url": "https://example.gov.cn/ev-01", "published_at": "2025-06-01", "added_at": "2025-06-02T00:00:00Z"} for fact in task.key_facts],
|
|
events_present=["message.created", "report.version.created"],
|
|
)
|
|
score = score_report(task, candidate, directed_hits=[True, True, True], skip={"template_adherence"})
|
|
assert score.metric("factual_accuracy").passed
|
|
assert score.metric("angle_coverage").passed
|
|
assert score.metric("citation_coverage").passed
|