## 新增功能 - 文件管理模块:分类树 + 文件上传/下载/删除 - 文件上传支持拖拽(Dragger)+ 手动上传(customRequest 模式) ## 页面布局统一(参照评测执行页) - 仪表盘/评测对象/评测场景/评测报告 全部改为全高 flex 布局 - 统一内联页头样式(h2 + 竖线分隔 + 描述) - 表格撑满高度、overflow 处理 - 每页添加刷新按钮 ## Bug 修复 - 分类树操作按钮 hover 不可见(CSS 规则缺失) - 文件上传失败(multipart boundary 缺失) - LLM API 响应 content blocks 数组格式支持(_extract_content_from_api_response) - response_time_max_ms 被静默忽略(隐式规则传空 params) - 空 messages 导致 IndexError 崩溃 - poll_reply 异常中止整个 run(缺 try/catch) - engine finally 未关闭 session - 3 个页面 UTC 时间戳解析偏差 8 小时 ## 后端 - EvalEngine: poll_reply 异常保护、空 dialog 保护、session 关闭 - LLM API 响应解析支持 content-block-array 格式 - 隐式 response_time 规则正确传递 max_ms 参数 ## 前端 - api.ts: 移除手动 Content-Type(让浏览器自动添加 boundary) - Files.tsx: customRequest 替代 beforeUpload、布局优化 - index.css: 分类树 hover 规则 - Targets/Scenarios/Home/Reports: 全高布局改造 - 3 个页面时间戳改用 formatDateTime()(修复 UTC 偏差) Co-Authored-By: Claude <noreply@anthropic.com>
97 lines
2.8 KiB
Python
97 lines
2.8 KiB
Python
"""
|
|
OpenClaw Skill example for AgentEvalTool.
|
|
|
|
This skill demonstrates how OpenClaw can call the AgentEvalTool CLI
|
|
to execute an evaluation run and fetch the report.
|
|
|
|
In OpenClaw, register this as a skill and configure the following parameters:
|
|
- target_id: ID of the registered evaluation target
|
|
- scenario_id: ID of the evaluation scenario
|
|
- report_format: "json" or "html" (default "json")
|
|
|
|
The skill assumes that the `agenteval` CLI is available on the system PATH.
|
|
"""
|
|
|
|
import json
|
|
import subprocess
|
|
from typing import Any
|
|
|
|
|
|
class AgentEvalSkill:
|
|
"""OpenClaw skill wrapper for AgentEvalTool."""
|
|
|
|
def __init__(self, config: dict[str, Any]):
|
|
self.config = config
|
|
|
|
def run(self) -> dict[str, Any]:
|
|
target_id = self.config["target_id"]
|
|
scenario_id = self.config["scenario_id"]
|
|
report_format = self.config.get("report_format", "json")
|
|
|
|
# 1. Trigger evaluation run
|
|
run_cmd = [
|
|
"agenteval",
|
|
"run",
|
|
"start",
|
|
"--target-id",
|
|
target_id,
|
|
"--scenario-id",
|
|
scenario_id,
|
|
]
|
|
run_result = subprocess.run(run_cmd, capture_output=True, text=True, check=False)
|
|
if run_result.returncode != 0:
|
|
return {
|
|
"ok": False,
|
|
"error": f"evaluation run failed: {run_result.stderr}",
|
|
"stdout": run_result.stdout,
|
|
}
|
|
|
|
# Extract run_id from CLI output (last line contains "run_id=xxx")
|
|
run_id = None
|
|
for line in reversed(run_result.stdout.strip().splitlines()):
|
|
if "run_id=" in line:
|
|
run_id = line.split("run_id=")[-1].strip().split()[0]
|
|
break
|
|
|
|
if not run_id:
|
|
return {
|
|
"ok": False,
|
|
"error": "could not extract run_id from CLI output",
|
|
"stdout": run_result.stdout,
|
|
}
|
|
|
|
# 2. Fetch report
|
|
report_cmd = [
|
|
"agenteval",
|
|
"report",
|
|
"show",
|
|
run_id,
|
|
"--format",
|
|
report_format,
|
|
]
|
|
report_result = subprocess.run(report_cmd, capture_output=True, text=True, check=False)
|
|
if report_result.returncode != 0:
|
|
return {
|
|
"ok": False,
|
|
"error": f"report fetch failed: {report_result.stderr}",
|
|
"run_id": run_id,
|
|
}
|
|
|
|
report_data = report_result.stdout
|
|
if report_format == "json":
|
|
try:
|
|
report_data = json.loads(report_data)
|
|
except json.JSONDecodeError:
|
|
pass
|
|
|
|
return {
|
|
"ok": True,
|
|
"run_id": run_id,
|
|
"report": report_data,
|
|
}
|
|
|
|
|
|
# Example entrypoint for OpenClaw runtime.
|
|
def execute(config: dict[str, Any]) -> dict[str, Any]:
|
|
return AgentEvalSkill(config).run()
|