39 lines
1.8 KiB
Python
39 lines
1.8 KiB
Python
"""Blind comparison between a collaboration report and a roundtable export."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from app.report_collaboration.eval.schema import ComparisonResult, EvalTask, ReportCandidate, TaskScore
|
|
from app.report_collaboration.eval.scorer import score_report
|
|
|
|
|
|
def compare_blind(
|
|
task: EvalTask,
|
|
collaboration: ReportCandidate,
|
|
roundtable: ReportCandidate,
|
|
*,
|
|
directed_hits: list[bool] | None = None,
|
|
) -> ComparisonResult:
|
|
left = score_report(task, collaboration.model_copy(update={"source": "collaboration"}), directed_hits=directed_hits)
|
|
right = score_report(task, roundtable.model_copy(update={"source": "roundtable"}), directed_hits=directed_hits)
|
|
deltas = {item.name: round(item.score - right.metric(item.name).score, 4) for item in left.metrics}
|
|
if left.composite > right.composite + 1e-9:
|
|
winner = "collaboration"
|
|
elif right.composite > left.composite + 1e-9:
|
|
winner = "roundtable"
|
|
else:
|
|
winner = "tie"
|
|
notes = _notes(winner, left, right, deltas)
|
|
return ComparisonResult(task_id=task.id, winner=winner, collaboration=left, roundtable=right, deltas=deltas, notes=notes)
|
|
|
|
|
|
def _notes(winner: str, left: TaskScore, right: TaskScore, deltas: dict[str, float]) -> str:
|
|
drivers = [name for name, delta in sorted(deltas.items(), key=lambda item: item[1], reverse=True) if delta > 0][:3]
|
|
if winner == "collaboration":
|
|
return "协作报告在 " + "、".join(drivers or ["综合分"]) + " 上优于会商导出;上线不得仅以「已接入 AgentScope」为由。"
|
|
if winner == "roundtable":
|
|
return "会商导出综合分更高。按决策门,本任务不能作为协作方案上线依据。"
|
|
return f"综合分持平({left.composite} vs {right.composite})。"
|
|
|
|
|
|
__all__ = ["compare_blind"]
|