deerflow-code/offline-backend-20260512/backend/app/report_collaboration/eval/thresholds.py
2026-09-07 18:24:55 +08:00

60 lines
1.8 KiB
Python

"""Pass gates and the runtime profile that the 20-task set was tuned against.
These numbers are the launch bar from the backend plan §11.2. They are not
model-judged: a canned gold report must clear them, and a known failure must not.
"""
from __future__ import annotations
from deerflow.config.report_collaboration_config import ReportCollaborationConfig
METRIC_GATES: dict[str, float] = {
"factual_accuracy": 0.85,
"source_quality": 0.80,
"citation_coverage": 0.85,
"angle_coverage": 1.00,
"logical_consistency": 1.00,
"template_adherence": 1.00,
"directed_modification": 0.80,
"time_cost": 1.00,
}
# Production knobs that keep the gold set green. Changing a default requires
# re-running `tests/test_report_collaboration_eval.py` and updating this table.
RECOMMENDED_RUNTIME_PROFILE: dict[str, object] = {
"enabled": False,
"worker_enabled": False,
"max_team_members": 8,
"max_parallel_tasks": 4,
"max_repair_rounds": 3,
"max_node_attempts": 3,
"max_react_iterations": 8,
"source_policy": "cite_required",
"token_budget": 250_000,
"max_model_calls": 200,
"max_retrieval_calls": 40,
"max_source_bytes": 2_000_000,
"max_concurrent_runs_per_user": 2,
"intent_confirmation_threshold": 0.8,
}
USAGE_CAPS = {
"duration_seconds": 7200.0,
"model_calls": 200,
"retrieval_calls": 40,
"input_tokens": 200_000,
"output_tokens": 50_000,
"cost": 0.0,
}
REQUIRED_EVENTS = ("message.created", "report.version.created")
def recommended_config() -> ReportCollaborationConfig:
return ReportCollaborationConfig.model_validate(
{key: value for key, value in RECOMMENDED_RUNTIME_PROFILE.items() if key in ReportCollaborationConfig.model_fields}
)
__all__ = ["METRIC_GATES", "RECOMMENDED_RUNTIME_PROFILE", "REQUIRED_EVENTS", "USAGE_CAPS", "recommended_config"]