## 核心变更
### 规则层全面异步化(DEBT-1)
- EvalRule.evaluate() 签名改为 async def,全量同步改造(无兼容层)
- LlmScoreRule._call_llm: requests.post → httpx.AsyncClient,彻底消除事件循环阻塞
- engine._save_rule_results: rule.evaluate() → await rule.evaluate()
### 工具函数去重(DEBT-2)
- 新建 agenteval/utils/llm.py,统一三个函数:
- extract_reply_text (原 5 处重复)
- extract_content_from_llm_response (原 2 处重复)
- parse_json_from_llm_text (统一 LLM 输出 JSON 解析)
- engine.py / llm_score.py / runs.py / report.py 全部切换到 utils.llm
### HTTP 通用通道(S1-3)
- 新建 channels/http.py (HttpChannel)
- 配置化 send_url / reply_url 模板 ({message}, {msg_id} 占位)
- dot-path 提取 msg_id 和 reply_text
- 可选 reply_ready_path 就绪标志
- 长连接 AsyncClient 复用
- ChannelFactory 注册 ChannelType.HTTP → HttpChannel
### 测试
- 新增 tests/unit/test_http_channel_and_rules.py (19 个测试)
- _get_path / health_check / send / poll_reply / 超时 / 就绪标志 / async 规则评估
- 测试总数:24 → 43,全部通过
Co-Authored-By: Claude <noreply@anthropic.com>
102 lines
3.7 KiB
Python
102 lines
3.7 KiB
Python
"""LLM-based scoring evaluation rule."""
|
||
|
||
import json
|
||
|
||
import httpx
|
||
|
||
from agenteval.evaluation.rules.base import EvalRule, RuleResult, register_rule
|
||
from agenteval.models import Case, Turn
|
||
from agenteval.utils.llm import extract_content_from_llm_response, extract_reply_text, parse_json_from_llm_text
|
||
|
||
|
||
@register_rule
|
||
class LlmScoreRule(EvalRule):
|
||
"""Use an external LLM to score reply quality against criteria."""
|
||
|
||
name = "llm_score"
|
||
|
||
async def evaluate(self, case: Case, dialog: list[Turn]) -> RuleResult:
|
||
if not dialog:
|
||
return RuleResult(passed=False, reason="无回复记录")
|
||
|
||
last_turn = dialog[-1]
|
||
reply_text = extract_reply_text(last_turn.reply)
|
||
|
||
question_text = ""
|
||
if len(dialog) >= 2:
|
||
question_text = extract_reply_text(dialog[-2].reply) or ""
|
||
if not question_text and last_turn.sent_message:
|
||
body = last_turn.sent_message.get("msgBody", "")
|
||
if isinstance(body, dict):
|
||
question_text = body.get("content", "")
|
||
else:
|
||
try:
|
||
question_text = json.loads(body).get("content", "")
|
||
except Exception:
|
||
question_text = str(body)
|
||
|
||
criteria = self.params.get("criteria", "")
|
||
min_score = float(self.params.get("min_score", 7))
|
||
api_url = self.params.get("api_url")
|
||
api_key = self.params.get("api_key")
|
||
model = self.params.get("model", "gpt-4o-mini")
|
||
|
||
if not api_url:
|
||
return RuleResult(passed=False, reason="LLM 评分规则未配置 api_url")
|
||
|
||
score, reason = await self._call_llm(api_url, api_key, model, question_text, reply_text, criteria)
|
||
if score is None:
|
||
return RuleResult(passed=False, reason=f"LLM 评分失败: {reason}")
|
||
|
||
passed = score >= min_score
|
||
return RuleResult(
|
||
passed=passed,
|
||
score=score / 10.0,
|
||
reason=f"LLM 评分 {score}/10,{'通过' if passed else '未通过'} (阈值 {min_score})",
|
||
)
|
||
|
||
async def _call_llm(
|
||
self,
|
||
api_url: str,
|
||
api_key: str | None,
|
||
model: str,
|
||
question: str,
|
||
reply: str,
|
||
criteria: str,
|
||
) -> tuple[float | None, str]:
|
||
"""Call the configured LLM API and parse a numeric score between 0 and 10."""
|
||
system_prompt = (
|
||
"你是一位严格的智能客服质量评估专家。请根据用户问题和智能体回复,"
|
||
f"按照以下标准打分(0-10分,10分最高):{criteria}\n"
|
||
'只输出一个 JSON 对象:{"score": number, "reason": "简短说明"}'
|
||
)
|
||
user_prompt = f"用户问题:{question}\n智能体回复:{reply}"
|
||
|
||
headers = {"Content-Type": "application/json"}
|
||
if api_key:
|
||
headers["Authorization"] = f"Bearer {api_key}"
|
||
|
||
payload = {
|
||
"model": model,
|
||
"messages": [
|
||
{"role": "system", "content": system_prompt},
|
||
{"role": "user", "content": user_prompt},
|
||
],
|
||
"temperature": 0.2,
|
||
}
|
||
|
||
try:
|
||
async with httpx.AsyncClient(timeout=60) as client:
|
||
resp = await client.post(api_url, headers=headers, json=payload)
|
||
resp.raise_for_status()
|
||
content = extract_content_from_llm_response(resp.json())
|
||
if not content:
|
||
return None, "LLM 返回内容为空"
|
||
|
||
parsed = parse_json_from_llm_text(content)
|
||
score = float(parsed["score"])
|
||
reason = parsed.get("reason", "")
|
||
return max(0.0, min(10.0, score)), reason
|
||
except Exception as exc:
|
||
return None, str(exc)
|