AgentEvalTool/backend/agenteval/evaluation/report_render.py
sinohqb 2484c207af feat(exploration): findings flow into report, analysis and export
Exploration sessions aggregate into a single exploration summary
(session counts, goal-achievement rate, issue lists from experience
records, judge conclusions when reviewed) that feeds three exits:
the campaign report gains an exploration dimension, the v0.7 analysis
stage-two input gains the summary (stats only, never full dialogues),
and the Markdown export appends a findings appendix after analysis and
comparison. With no exploration data every output stays unchanged.
2026-08-03 19:16:33 +08:00

459 lines
18 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""Pure renderers: report dict in, HTML/Markdown/JSON string out.
No I/O and no storage access — every function takes the dict produced by the
generation side (``evaluation.report``) so rendering can be unit-tested from a
hand-built dict. Report *content* decisions (which cases pass, which rates to
show) belong to generation; this module only formats.
"""
import json
from datetime import datetime
from typing import Any, Optional
from jinja2 import Template
HTML_TEMPLATE = """<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8">
<title>评测报告 - {{ report.run_id }}</title>
<style>
body { font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif; margin: 40px; background: #f5f5f5; }
.container { max-width: 960px; margin: 0 auto; background: #fff; padding: 32px; border-radius: 8px; box-shadow: 0 2px 8px rgba(0,0,0,0.05); }
h1 { margin-top: 0; }
.summary { display: grid; grid-template-columns: repeat(auto-fit, minmax(140px, 1fr)); gap: 16px; margin: 24px 0; }
.card { background: #fafafa; border-radius: 6px; padding: 16px; text-align: center; }
.card .value { font-size: 24px; font-weight: 700; }
.card .label { color: #666; font-size: 14px; margin-top: 4px; }
.case { border: 1px solid #e0e0e0; border-radius: 6px; margin: 16px 0; padding: 16px; }
.case-title { font-weight: 600; margin-bottom: 8px; }
.turn { background: #f9f9f9; border-radius: 4px; padding: 12px; margin: 8px 0; }
.message-label { color: #666; font-size: 12px; }
.rule { display: flex; align-items: center; gap: 8px; margin: 6px 0; }
.badge { padding: 2px 8px; border-radius: 4px; font-size: 12px; }
.pass { background: #e6f7e6; color: #2e7d32; }
.fail { background: #ffebee; color: #c62828; }
pre { white-space: pre-wrap; word-break: break-word; background: #f5f5f5; padding: 8px; border-radius: 4px; }
</style>
</head>
<body>
<div class="container">
<h1>评测报告</h1>
<p>评测对象:{{ report.target_name }}{{ report.target_id }}</p>
<p>评测场景:{{ report.scenario_name }}{{ report.scenario_id }}</p>
<p>执行时间:{{ report.started_at }} 至 {{ report.completed_at or '进行中' }}</p>
<div class="summary">
<div class="card">
<div class="value">{{ report.summary.total_cases }}</div>
<div class="label">用例总数</div>
</div>
<div class="card">
<div class="value">{{ report.summary.passed_cases }}</div>
<div class="label">通过用例</div>
</div>
<div class="card">
<div class="value">{{ report.summary.total_rules }}</div>
<div class="label">规则总数</div>
</div>
<div class="card">
<div class="value">{{ "%.2f"|format(report.summary.pass_rate * 100) }}%</div>
<div class="label">规则通过率</div>
</div>
</div>
{% for case in report.cases %}
<div class="case">
<div class="case-title">用例 {{ case.case_id }}</div>
{% for turn in case.turns %}
<div class="turn">
<div class="message-label">用户消息</div>
<pre>{{ turn.sent_text }}</pre>
<div class="message-label">智能体回复({{ turn.latency_ms }}ms</div>
<pre>{{ turn.reply_text or '(无回复)' }}</pre>
</div>
{% endfor %}
<div>
{% for result in case.results %}
<div class="rule">
<span class="badge {{ 'pass' if result.passed else 'fail' }}">{{ '通过' if result.passed else '失败' }}</span>
<span>{{ result.rule_type }}: {{ result.reason }}</span>
</div>
{% endfor %}
</div>
</div>
{% endfor %}
</div>
</body>
</html>
"""
def _pct(v: Optional[float]) -> str:
return "" if v is None else f"{v * 100:.1f}%"
def _ms(v: Optional[float]) -> str:
return "" if v is None else f"{v:.0f}ms"
def render_html(report: dict[str, Any]) -> str:
"""Render a run report dict as an HTML string."""
return Template(HTML_TEMPLATE).render(report=report)
def render_json(report: dict[str, Any]) -> str:
"""Render a report dict as a JSON string."""
return json.dumps(report, ensure_ascii=False, indent=2)
def render_markdown(report: dict[str, Any]) -> str:
"""Render a run report dict as Markdown."""
s = report["summary"]
judged_rate = s.get("judged_pass_rate")
judged_rate_text = "" if judged_rate is None else f"{judged_rate * 100:.1f}%"
lines: list[str] = [
f"# 评测报告 — {report.get('scenario_name', report.get('run_id', ''))}",
"",
f"**评测对象**: {report.get('target_name', '-')} ",
f"**评测场景**: {report.get('scenario_name', '-')} ",
f"**状态**: {report.get('status', '-')} ",
f"**开始时间**: {report.get('started_at', '-')} ",
f"**完成时间**: {report.get('completed_at', '-')} ",
"",
"## 汇总",
"",
"| 指标 | 数值 |",
"|------|------|",
f"| 总用例数 | {s['total_cases']} |",
f"| 通过用例 | {s['passed_cases']} |",
f"| 失败用例 | {s['failed_cases']} |",
f"| 总规则数 | {s['total_rules']} |",
f"| 通过规则 | {s['passed_rules']} |",
f"| 通过率 | {s['pass_rate'] * 100:.1f}% |",
f"| 连通用例 | {s.get('connectivity_cases', 0)} |",
f"| 判定型通过率 | {judged_rate_text} |",
"",
"## 用例明细",
"",
]
for case in report.get("cases", []):
if case.get("connectivity"):
badge = "🔗"
else:
badge = "" if case.get("passed") else ""
title = f"### {badge} 用例 `{case['case_id']}`"
if case.get("connectivity"):
title += "(连通用例,未配置判定标准)"
lines.append(title)
lines.append("")
for turn in case.get("turns", []):
lines.append(f"**第 {turn['round']} 轮**")
lines.append("")
lines.append(f"> **用户**: {turn.get('sent_text', '')}")
lines.append("")
reply = turn.get("reply_text") or "(无回复)"
lines.append(f"> **智能体**: {reply}")
if turn.get("latency_ms") is not None:
lines.append(f"> *延迟: {turn['latency_ms']}ms*")
lines.append("")
if case.get("results"):
lines.append("**规则评估结果**")
lines.append("")
lines.append("| 规则 | 结果 | 评分 | 说明 |")
lines.append("|------|------|------|------|")
for r in case["results"]:
badge = "" if r["passed"] else ""
score = f"{r['score']:.2f}" if r.get("score") is not None else "-"
lines.append(f"| {r['rule_type']} | {badge} | {score} | {r.get('reason', '')} |")
lines.append("")
return "\n".join(lines)
_SEVERITY_LABELS = {"high": "", "medium": "", "low": ""}
def _render_analysis_lines(analysis: dict[str, Any], scenario_names: dict[str, str]) -> list[str]:
"""Appendix for the campaign export: the structured analysis blocks."""
def _name(sid: str) -> str:
return scenario_names.get(sid, sid[:8])
lines = ["", "## 智能分析", "", "### 总体结论", "", str(analysis.get("overall") or ""), ""]
problems = analysis.get("problems") or []
if problems:
lines += ["### 问题诊断", ""]
for p in problems:
severity = _SEVERITY_LABELS.get(p.get("severity"), "")
names = "".join(_name(sid) for sid in p.get("scenario_ids") or [])
head = f"- **[{severity}] {p.get('title', '')}**"
if names:
head += f"(场景:{names}"
lines.append(head)
if p.get("description"):
lines.append(f" {p['description']}")
evidence = p.get("evidence_run_ids") or []
if evidence:
lines.append(" 证据 run" + " ".join(f"`{rid}`" for rid in evidence))
lines.append("")
narratives = analysis.get("scenario_narratives") or []
if narratives:
lines += ["### 分场景叙述", ""]
for n in narratives:
lines.append(f"- **{_name(str(n.get('scenario_id', '')))}**{n.get('narrative', '')}")
lines.append("")
suggestions = sorted(analysis.get("suggestions") or [], key=lambda s: s.get("priority", 0))
if suggestions:
lines += ["### 改善建议", ""]
lines += [f"{i}. {s.get('text', '')}" for i, s in enumerate(suggestions, 1)]
lines.append("")
return lines
_CAMPAIGN_STATUS_LABELS = {
"planned": "计划中", "running": "进行中", "completed": "已完成",
"cancelled": "已取消", "failed": "失败",
}
_TREND_LABELS = {"improving": "改善", "stable": "平稳", "regressing": "退化"}
_EVOLUTION_LABELS = {"new": "新增", "persisting": "持续", "resolved": "消解"}
_TRACKING_LABELS = {"addressed": "已落实", "partial": "部分落实", "unaddressed": "未落实", "new": "新增"}
def _offset(seconds: float) -> str:
"""Human-readable window offset: 86400 → 24h, 1800 → 30m, 45 → 45s."""
s = int(round(seconds))
if s % 3600 == 0:
return f"{s // 3600}h"
if s % 60 == 0:
return f"{s // 60}m"
return f"{s}s"
def _dt(iso: Optional[str]) -> str:
"""ISO timestamp → `2026-08-01 08:00`; unparseable input passes through."""
if not iso:
return "-"
try:
return datetime.fromisoformat(iso.replace("Z", "+00:00")).strftime("%Y-%m-%d %H:%M")
except ValueError:
return iso
def _diff_cell(metric: str, pair: dict[str, Any]) -> str:
"""One metric cell of the diff table: `基线 → 本期±delta`."""
def _val(v: Optional[float]) -> str:
if v is None:
return ""
return _ms(v) if metric == "avg_latency_ms" else _pct(v)
delta = pair.get("delta")
if delta is None:
delta_text = ""
elif metric == "avg_latency_ms":
delta_text = f"{delta:+.1f}ms"
else:
delta_text = f"{delta * 100:+.1f}pp"
return f"{_val(pair.get('baseline'))}{_val(pair.get('current'))}{delta_text}"
def _render_comparison_lines(comparison: dict[str, Any], scenario_names: dict[str, str]) -> list[str]:
"""Appendix for the campaign export: narrative + mechanical metric diff.
``comparison`` carries the completed narrative result plus context resolved
on the read path (baseline name/times, model name, generated-at, diff).
"""
def _name(sid: str) -> str:
return scenario_names.get(sid, sid[:8])
result = comparison.get("result") or {}
lines = ["", "## 周期对比", ""]
meta: list[str] = []
if comparison.get("baseline_name"):
base = f"基线:「{comparison['baseline_name']}"
if comparison.get("baseline_completed_at"):
base += f"(完成于 {_dt(comparison['baseline_completed_at'])}"
meta.append(base)
if comparison.get("model_name"):
meta.append(f"分析模型:{comparison['model_name']}")
if comparison.get("updated_at"):
meta.append(f"生成于:{_dt(comparison['updated_at'])}")
if meta:
lines += [" · ".join(meta), ""]
trend = _TREND_LABELS.get(result.get("trend"), "平稳")
lines += [f"**趋势**{trend}{result.get('summary') or ''}", ""]
diff = comparison.get("metric_diff")
if diff:
lines += [
"### 指标变化",
"",
"| 维度 | 通过率 | 可用性 | 平均时延 |",
"|------|------|------|------|",
]
def _row(label: str, block: dict[str, Any]) -> str:
return (
f"| {label} | {_diff_cell('pass_rate', block['pass_rate'])} | "
f"{_diff_cell('availability', block['availability'])} | "
f"{_diff_cell('avg_latency_ms', block['avg_latency_ms'])} |"
)
lines.append(_row("整窗(总体)", diff["overall"]))
for s in diff.get("scenarios") or []:
label = s.get("scenario_name") or _name(s.get("scenario_id", ""))
lines.append(_row(label, s))
lines.append("")
evolution = result.get("problem_evolution") or []
if evolution:
lines += ["### 问题演变", ""]
for p in evolution:
status = _EVOLUTION_LABELS.get(p.get("status"), "持续")
names = "".join(_name(sid) for sid in p.get("scenario_ids") or [])
head = f"- **[{status}] {p.get('title', '')}**"
if names:
head += f"(场景:{names}"
lines.append(head)
if p.get("detail"):
lines.append(f" {p['detail']}")
lines.append("")
tracking = result.get("suggestion_tracking") or []
if tracking:
lines += ["### 建议落实情况", ""]
for t in tracking:
status = _TRACKING_LABELS.get(t.get("status"), "未落实")
lines.append(f"- **[{status}] {t.get('text', '')}**")
if t.get("note"):
lines.append(f" {t['note']}")
lines.append("")
return lines
def _window_line(report: dict[str, Any]) -> str:
"""Human-readable window line with the 正式线 / 加速调试线 wording."""
window = report.get("window_seconds") or 0
scale = float(report.get("time_scale") or 1)
if scale == 1:
return f"**窗口**: {_offset(window)}(正式线)"
wall = f",压缩后实际耗时约 {_offset(window / scale)}"
return f"**窗口**: {_offset(window)}(加速调试线 ×{scale:g}{wall}"
def _render_exploration_lines(exploration: dict[str, Any]) -> list[str]:
"""探索发现附录:会话统计 + 问题清单 + judge 复核发现(有才渲染)。"""
lines: list[str] = [
"",
"## 探索发现",
"",
"| 指标 | 数值 |",
"|------|------|",
f"| 探索会话数 | {exploration.get('session_count', 0)} |",
f"| 有体验记录会话数 | {exploration.get('sessions_with_experience', 0)} |",
f"| 目标达成率 | {_pct(exploration.get('goal_achievement_rate'))} |",
"",
"### 问题清单",
"",
]
issues = exploration.get("issues") or []
misled = exploration.get("misled") or []
if not issues and not misled:
lines.append("")
else:
for item in issues:
lines.append(f"- {item['issue']} ×{item['count']}")
for item in misled:
lines.append(f"- (被误导){item['issue']} ×{item['count']}")
judge = exploration.get("judge_review")
if judge:
lines += [
"",
"### judge 复核",
"",
f"已复核 {judge.get('reviewed_sessions', 0)} 个会话:",
]
findings = judge.get("findings") or []
if findings:
for item in findings:
lines.append(f"- [{item.get('dimension')}/{item.get('rating')}] {item.get('comment')}")
else:
lines.append("- 未发现问题")
for summary_text in judge.get("summaries") or []:
lines.append(f"- 复核结论:{summary_text}")
return lines
def render_campaign_markdown(
report: dict[str, Any],
*,
analysis: Optional[dict[str, Any]] = None,
comparison: Optional[dict[str, Any]] = None,
exploration: Optional[dict[str, Any]] = None,
target_name: Optional[str] = None,
scenario_names: Optional[dict[str, str]] = None,
) -> str:
"""Render a dual-axis campaign report dict as Markdown.
``analysis`` and ``comparison`` are the stored 智能分析 / 周期对比 results
(completed only); ``exploration`` is the on-the-fly 探索发现 aggregate;
when absent the corresponding appendix is omitted entirely (缺则无痕).
"""
s = report["summary"]
status = _CAMPAIGN_STATUS_LABELS.get(report.get("status"), report.get("status") or "-")
target = target_name or report.get("target_id") or "-"
lines: list[str] = [
f"# 活动周期报告 — {report['name']}",
"",
f"**评测对象**: {target} ",
f"**状态**: {status} ",
_window_line(report) + " ",
f"**开始时间**: {_dt(report.get('started_at'))} ",
f"**完成时间**: {_dt(report.get('completed_at'))} ",
"",
"## 汇总",
"",
"| 指标 | 数值 |",
"|------|------|",
f"| 子运行总数 | {s['total_runs']} |",
f"| 已完成 | {s['completed_runs']} |",
f"| 整窗通过率 | {_pct(s['overall_pass_rate'])} |",
f"| 整窗可用性 | {_pct(s['overall_availability'])} |",
f"| 平均时延 | {_ms(s['avg_latency_ms'])} |",
"",
"## 时间趋势",
"",
"| 时段 | 运行数 | 通过率 | 可用性 | 时延 |",
"|------|------|------|------|------|",
]
for b in report["time_trend"]:
lines.append(
f"| {_offset(b['start_seconds'])}{_offset(b['end_seconds'])} | {b['run_count']} | "
f"{_pct(b['pass_rate'])} | {_pct(b['availability'])} | {_ms(b['avg_latency_ms'])} |"
)
lines += [
"",
"## 能力汇总",
"",
"| 场景 | 运行数 | 通过率 | 可用性 | 时延 |",
"|------|------|------|------|------|",
]
for c in report["capability_summary"]:
lines.append(
f"| {c['scenario_name']} | {c['run_count']} | {_pct(c['pass_rate'])} | "
f"{_pct(c['availability'])} | {_ms(c['avg_latency_ms'])} |"
)
if analysis:
lines += _render_analysis_lines(analysis, scenario_names or {})
if comparison:
lines += _render_comparison_lines(comparison, scenario_names or {})
if exploration:
lines += _render_exploration_lines(exploration)
return "\n".join(lines)