## 新增功能 - 文件管理模块:分类树 + 文件上传/下载/删除 - 文件上传支持拖拽(Dragger)+ 手动上传(customRequest 模式) ## 页面布局统一(参照评测执行页) - 仪表盘/评测对象/评测场景/评测报告 全部改为全高 flex 布局 - 统一内联页头样式(h2 + 竖线分隔 + 描述) - 表格撑满高度、overflow 处理 - 每页添加刷新按钮 ## Bug 修复 - 分类树操作按钮 hover 不可见(CSS 规则缺失) - 文件上传失败(multipart boundary 缺失) - LLM API 响应 content blocks 数组格式支持(_extract_content_from_api_response) - response_time_max_ms 被静默忽略(隐式规则传空 params) - 空 messages 导致 IndexError 崩溃 - poll_reply 异常中止整个 run(缺 try/catch) - engine finally 未关闭 session - 3 个页面 UTC 时间戳解析偏差 8 小时 ## 后端 - EvalEngine: poll_reply 异常保护、空 dialog 保护、session 关闭 - LLM API 响应解析支持 content-block-array 格式 - 隐式 response_time 规则正确传递 max_ms 参数 ## 前端 - api.ts: 移除手动 Content-Type(让浏览器自动添加 boundary) - Files.tsx: customRequest 替代 beforeUpload、布局优化 - index.css: 分类树 hover 规则 - Targets/Scenarios/Home/Reports: 全高布局改造 - 3 个页面时间戳改用 formatDateTime()(修复 UTC 偏差) Co-Authored-By: Claude <noreply@anthropic.com>
209 lines
7.9 KiB
Python
209 lines
7.9 KiB
Python
"""Report generation for evaluation runs."""
|
||
|
||
import json
|
||
from datetime import datetime
|
||
from pathlib import Path
|
||
from typing import Any, Optional
|
||
|
||
from jinja2 import Template
|
||
|
||
from agenteval.storage.db import DATA_DIR
|
||
from agenteval.storage.repository import RunRepository, ScenarioRepository, TargetRepository
|
||
|
||
HTML_TEMPLATE = """<!DOCTYPE html>
|
||
<html lang="zh-CN">
|
||
<head>
|
||
<meta charset="UTF-8">
|
||
<title>评测报告 - {{ report.run_id }}</title>
|
||
<style>
|
||
body { font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, sans-serif; margin: 40px; background: #f5f5f5; }
|
||
.container { max-width: 960px; margin: 0 auto; background: #fff; padding: 32px; border-radius: 8px; box-shadow: 0 2px 8px rgba(0,0,0,0.05); }
|
||
h1 { margin-top: 0; }
|
||
.summary { display: grid; grid-template-columns: repeat(auto-fit, minmax(140px, 1fr)); gap: 16px; margin: 24px 0; }
|
||
.card { background: #fafafa; border-radius: 6px; padding: 16px; text-align: center; }
|
||
.card .value { font-size: 24px; font-weight: 700; }
|
||
.card .label { color: #666; font-size: 14px; margin-top: 4px; }
|
||
.case { border: 1px solid #e0e0e0; border-radius: 6px; margin: 16px 0; padding: 16px; }
|
||
.case-title { font-weight: 600; margin-bottom: 8px; }
|
||
.turn { background: #f9f9f9; border-radius: 4px; padding: 12px; margin: 8px 0; }
|
||
.message-label { color: #666; font-size: 12px; }
|
||
.rule { display: flex; align-items: center; gap: 8px; margin: 6px 0; }
|
||
.badge { padding: 2px 8px; border-radius: 4px; font-size: 12px; }
|
||
.pass { background: #e6f7e6; color: #2e7d32; }
|
||
.fail { background: #ffebee; color: #c62828; }
|
||
pre { white-space: pre-wrap; word-break: break-word; background: #f5f5f5; padding: 8px; border-radius: 4px; }
|
||
</style>
|
||
</head>
|
||
<body>
|
||
<div class="container">
|
||
<h1>评测报告</h1>
|
||
<p>评测对象:{{ report.target_name }}({{ report.target_id }})</p>
|
||
<p>评测场景:{{ report.scenario_name }}({{ report.scenario_id }})</p>
|
||
<p>执行时间:{{ report.started_at }} 至 {{ report.completed_at or '进行中' }}</p>
|
||
|
||
<div class="summary">
|
||
<div class="card">
|
||
<div class="value">{{ report.summary.total_cases }}</div>
|
||
<div class="label">用例总数</div>
|
||
</div>
|
||
<div class="card">
|
||
<div class="value">{{ report.summary.passed_cases }}</div>
|
||
<div class="label">通过用例</div>
|
||
</div>
|
||
<div class="card">
|
||
<div class="value">{{ report.summary.total_rules }}</div>
|
||
<div class="label">规则总数</div>
|
||
</div>
|
||
<div class="card">
|
||
<div class="value">{{ "%.2f"|format(report.summary.pass_rate * 100) }}%</div>
|
||
<div class="label">规则通过率</div>
|
||
</div>
|
||
</div>
|
||
|
||
{% for case in report.cases %}
|
||
<div class="case">
|
||
<div class="case-title">用例 {{ case.case_id }}</div>
|
||
{% for turn in case.turns %}
|
||
<div class="turn">
|
||
<div class="message-label">用户消息</div>
|
||
<pre>{{ turn.sent_text }}</pre>
|
||
<div class="message-label">智能体回复({{ turn.latency_ms }}ms)</div>
|
||
<pre>{{ turn.reply_text or '(无回复)' }}</pre>
|
||
</div>
|
||
{% endfor %}
|
||
<div>
|
||
{% for result in case.results %}
|
||
<div class="rule">
|
||
<span class="badge {{ 'pass' if result.passed else 'fail' }}">{{ '通过' if result.passed else '失败' }}</span>
|
||
<span>{{ result.rule_type }}: {{ result.reason }}</span>
|
||
</div>
|
||
{% endfor %}
|
||
</div>
|
||
</div>
|
||
{% endfor %}
|
||
</div>
|
||
</body>
|
||
</html>
|
||
"""
|
||
|
||
|
||
def _extract_text(data: Any) -> str:
|
||
if data is None:
|
||
return ""
|
||
if isinstance(data, str):
|
||
return data
|
||
if isinstance(data, dict):
|
||
body = data.get("msgBody") or data.get("content", "")
|
||
if isinstance(body, dict):
|
||
return body.get("content", "")
|
||
return str(body)
|
||
return str(data)
|
||
|
||
|
||
def generate_report(run_id: str, session=None) -> dict[str, Any]:
|
||
"""Build a structured report dict for a run."""
|
||
run_repo = RunRepository(session)
|
||
target_repo = TargetRepository(session)
|
||
scenario_repo = ScenarioRepository(session)
|
||
|
||
run = run_repo.get(run_id)
|
||
if not run:
|
||
raise ValueError(f"run not found: {run_id}")
|
||
|
||
target = target_repo.get(run.target_id)
|
||
scenario = scenario_repo.get(run.scenario_id)
|
||
|
||
turns = run_repo.get_turns(run_id)
|
||
results = run_repo.get_results(run_id)
|
||
|
||
# Group by case
|
||
case_map: dict[str, dict[str, Any]] = {}
|
||
for turn in turns:
|
||
case_map.setdefault(turn.case_id, {"turns": [], "results": []})
|
||
sent = turn.get_sent_message()
|
||
reply = turn.get_reply()
|
||
case_map[turn.case_id]["turns"].append(
|
||
{
|
||
"round": turn.round_index,
|
||
"sent_text": _extract_text(sent.get("msgBody")),
|
||
"reply_text": _extract_text(reply.get("msgBody") if reply else None),
|
||
"latency_ms": turn.latency_ms,
|
||
"question_msg_id": turn.question_msg_id,
|
||
}
|
||
)
|
||
|
||
for result in results:
|
||
case_map.setdefault(result.case_id, {"turns": [], "results": []})
|
||
case_map[result.case_id]["results"].append(
|
||
{
|
||
"rule_type": result.rule_type,
|
||
"passed": result.passed,
|
||
"score": result.score,
|
||
"reason": result.reason,
|
||
}
|
||
)
|
||
|
||
cases = []
|
||
for case_id in sorted(case_map.keys()):
|
||
item = case_map[case_id]
|
||
cases.append(
|
||
{
|
||
"case_id": case_id,
|
||
"turns": sorted(item["turns"], key=lambda x: x["round"]),
|
||
"results": item["results"],
|
||
}
|
||
)
|
||
|
||
summary = run.summary or {}
|
||
return {
|
||
"run_id": run.id,
|
||
"target_id": run.target_id,
|
||
"target_name": target.name if target else "未知",
|
||
"scenario_id": run.scenario_id,
|
||
"scenario_name": scenario.name if scenario else "未知",
|
||
"status": run.status.value,
|
||
"started_at": run.started_at.isoformat() if run.started_at else None,
|
||
"completed_at": run.completed_at.isoformat() if run.completed_at else None,
|
||
"summary": {
|
||
"total_cases": summary.get("total_cases", 0),
|
||
"passed_cases": summary.get("passed_cases", 0),
|
||
"failed_cases": summary.get("failed_cases", 0),
|
||
"total_rules": summary.get("total_rules", 0),
|
||
"passed_rules": summary.get("passed_rules", 0),
|
||
"pass_rate": summary.get("pass_rate", 0.0),
|
||
},
|
||
"cases": cases,
|
||
}
|
||
|
||
|
||
def render_json_report(run_id: str, session=None) -> str:
|
||
"""Render a report as JSON string."""
|
||
report = generate_report(run_id, session)
|
||
return json.dumps(report, ensure_ascii=False, indent=2)
|
||
|
||
|
||
def render_html_report(run_id: str, session=None) -> str:
|
||
"""Render a report as HTML string."""
|
||
report = generate_report(run_id, session)
|
||
template = Template(HTML_TEMPLATE)
|
||
return template.render(report=report)
|
||
|
||
|
||
def save_report(run_id: str, fmt: str = "html", output_dir: Optional[Path] = None) -> Path:
|
||
"""Generate and save a report to disk."""
|
||
output_dir = output_dir or DATA_DIR / "reports"
|
||
output_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
timestamp = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||
if fmt == "html":
|
||
content = render_html_report(run_id)
|
||
path = output_dir / f"report_{run_id}_{timestamp}.html"
|
||
elif fmt == "json":
|
||
content = render_json_report(run_id)
|
||
path = output_dir / f"report_{run_id}_{timestamp}.json"
|
||
else:
|
||
raise ValueError(f"unsupported report format: {fmt}")
|
||
|
||
path.write_text(content, encoding="utf-8")
|
||
return path
|