"""LLM-based scoring evaluation rule.""" import json from typing import Any import requests from agenteval.evaluation.rules.base import EvalRule, RuleResult, register_rule from agenteval.models import Case, Turn def _extract_text(reply: Any) -> str: if reply is None: return "" if isinstance(reply, str): return reply if isinstance(reply, dict): body = reply.get("msgBody") or reply.get("content", "") if isinstance(body, dict): return body.get("content", "") return str(body) return str(reply) def _extract_content_from_api_response(data: dict) -> str: """Extract text content from an LLM API response. Handles both OpenAI format (choices[0].message.content as string) and content-block-array format used by Anthropic-compatible APIs (choices[0].message.content as list of {type, text/text} blocks). """ try: content = data["choices"][0]["message"]["content"] except (KeyError, IndexError, TypeError): return "" if isinstance(content, str): return content if isinstance(content, list): parts = [] for block in content: if not isinstance(block, dict): continue if block.get("type") == "text": parts.append(block.get("text") or block.get("content") or "") return "\n".join(parts) return str(content) @register_rule class LlmScoreRule(EvalRule): """Use an external LLM to score reply quality against criteria.""" name = "llm_score" def evaluate(self, case: Case, dialog: list[Turn]) -> RuleResult: if not dialog: return RuleResult(passed=False, reason="无回复记录") last_turn = dialog[-1] reply_text = _extract_text(last_turn.reply) question_text = "" if len(dialog) >= 2: question_text = _extract_text(dialog[-2].reply) or "" if not question_text and last_turn.sent_message: body = last_turn.sent_message.get("msgBody", "") if isinstance(body, dict): question_text = body.get("content", "") else: try: parsed = json.loads(body) question_text = parsed.get("content", "") except Exception: question_text = str(body) criteria = self.params.get("criteria", "") min_score = float(self.params.get("min_score", 7)) api_url = self.params.get("api_url") api_key = self.params.get("api_key") model = self.params.get("model", "gpt-4o-mini") if not api_url: return RuleResult(passed=False, reason="LLM 评分规则未配置 api_url") score, reason = self._call_llm(api_url, api_key, model, question_text, reply_text, criteria) if score is None: return RuleResult(passed=False, reason=f"LLM 评分失败: {reason}") passed = score >= min_score return RuleResult( passed=passed, score=score / 10.0, reason=f"LLM 评分 {score}/10,{'通过' if passed else '未通过'} (阈值 {min_score})", ) def _call_llm( self, api_url: str, api_key: str | None, model: str, question: str, reply: str, criteria: str, ) -> tuple[float | None, str]: """Call the configured LLM API and parse a numeric score between 0 and 10.""" system_prompt = ( "你是一位严格的智能客服质量评估专家。请根据用户问题和智能体回复," f"按照以下标准打分(0-10分,10分最高):{criteria}\n" "只输出一个 JSON 对象:{\"score\": number, \"reason\": \"简短说明\"}" ) user_prompt = f"用户问题:{question}\n智能体回复:{reply}" headers = {"Content-Type": "application/json"} if api_key: headers["Authorization"] = f"Bearer {api_key}" payload = { "model": model, "messages": [ {"role": "system", "content": system_prompt}, {"role": "user", "content": user_prompt}, ], "temperature": 0.2, } try: resp = requests.post(api_url, headers=headers, json=payload, timeout=60) resp.raise_for_status() data = resp.json() content = _extract_content_from_api_response(data) if not content: return None, "LLM 返回内容为空" # Try to parse JSON from the content try: parsed = json.loads(content) except json.JSONDecodeError: # Fallback: extract JSON substring start = content.find("{") end = content.rfind("}") if start == -1 or end == -1: return None, "LLM 返回格式无法解析" parsed = json.loads(content[start : end + 1]) score = float(parsed["score"]) reason = parsed.get("reason", "") return max(0.0, min(10.0, score)), reason except Exception as exc: return None, str(exc)