"""Decision-log service (P3 deepening, S2). Pulled out of ``web/routers/intelligent_evals.py`` so the router only handles HTTP validation and error translation. The ORM writes and reads now live here. """ from typing import Any from sqlmodel import Session, select from agenteval.intelligent_eval.models import IntelligentEvalStatus from agenteval.storage.db import ( IntelligentEvalDB, IntelligentEvalDecisionLogDB, IntelligentEvalSessionDB, ) def _log_to_dict(log: IntelligentEvalDecisionLogDB) -> dict: return { "id": log.id, "eval_id": log.eval_id, "decision_type": log.decision_type, "reason": log.reason, "context": log.get_context(), "cron_id": log.cron_id, "created_at": log.created_at.isoformat() if log.created_at else None, } def _require_eval(eval_id: str, session: Session) -> None: eval_db = session.get(IntelligentEvalDB, eval_id) if eval_db is None or eval_db.status == IntelligentEvalStatus.DELETED.value: raise LookupError(f"intelligent eval {eval_id} not found") def _append_row( eval_id: str, decision_type: str, reason: str, cron_id: str, context: dict[str, Any], session: Session, ) -> dict: log = IntelligentEvalDecisionLogDB( eval_id=eval_id, decision_type=decision_type, reason=reason, cron_id=cron_id, ) log.set_context(context) session.add(log) session.commit() session.refresh(log) return _log_to_dict(log) def create_decision_log( eval_id: str, decision_type: str, reason: str, cron_id: str, context: dict[str, Any], session: Session, ) -> dict: """Agent-reporting entry: create a decision log, or return the existing one if the (eval_id, decision_type, context) tuple is already recorded. P3 真问题修复 (T8 / Gitea #6): agent 会在同一分钟内重复上报相同决策, 按 context JSON 去重保表干净。**去重只服务 agent 上报路径**——平台落账 每次都是新事实,用 ``append_decision_log``,调用方无需知道去重存在。 Raises ``LookupError`` if eval not found. """ _require_eval(eval_id, session) # Dedupe: same (eval, decision_type, context) → return existing. # Limit to 100 records to avoid loading too many into memory; in practice, # an eval rarely has more than a few dozen logs of the same type. for existing in session.exec( select(IntelligentEvalDecisionLogDB) .where( IntelligentEvalDecisionLogDB.eval_id == eval_id, IntelligentEvalDecisionLogDB.decision_type == decision_type, ) .limit(100) ).all(): if existing.get_context() == context: return _log_to_dict(existing) return _append_row(eval_id, decision_type, reason, cron_id, context, session) def append_decision_log( eval_id: str, decision_type: str, reason: str, cron_id: str, context: dict[str, Any], session: Session, ) -> dict: """Platform-bookkeeping entry: 纯追加,不去重。 平台每次落账都是新事实(attempt 递增、task_id 不同、闸门判定), 相同 context 也总是追加一行。 Raises ``LookupError`` if eval not found. """ _require_eval(eval_id, session) return _append_row(eval_id, decision_type, reason, cron_id, context, session) def count_decisions(eval_id: str, decision_type: str, session: Session) -> int: """该评估某类型决策日志的条数(平台闸门计数与 attempt 落账的计数原语)。""" return len( session.exec( select(IntelligentEvalDecisionLogDB).where( IntelligentEvalDecisionLogDB.eval_id == eval_id, IntelligentEvalDecisionLogDB.decision_type == decision_type, ) ).all() ) def list_decision_logs(eval_id: str, session: Session) -> list[dict]: """List decision logs for an eval. Raises ``LookupError`` if eval not found.""" _require_eval(eval_id, session) logs = session.exec( select(IntelligentEvalDecisionLogDB) .where(IntelligentEvalDecisionLogDB.eval_id == eval_id) .order_by(IntelligentEvalDecisionLogDB.created_at.desc()) ).all() return [_log_to_dict(log) for log in logs] def _supplement_executing( ev: IntelligentEvalDB, sessions: list, completed: int, estimated: int, types: set[str], session: Session, ) -> int: """Supplement decision logs for EXECUTING evals.""" added = 0 if "execute_session" not in types and completed < estimated: _append_row( ev.id, "execute_session", "平台兜底:时段欠账需执行会话", "platform", {"platform_supplemented": True, "completed": completed, "estimated": estimated}, session, ) added += 1 elif "start_analysis" not in types and sessions and completed >= estimated: _append_row( ev.id, "start_analysis", "平台兜底:所有会话已完成开始分析", "platform", {"platform_supplemented": True, "completed": completed, "estimated": estimated}, session, ) added += 1 return added def _supplement_completed( ev: IntelligentEvalDB, sessions: list, completed: int, estimated: int, plan: dict, types: set[str], session: Session, ) -> int: """Supplement decision logs for COMPLETED evals (historical backfill).""" added = 0 if "execute_session" not in types: slots = plan.get("time_distribution") or [] if slots: for slot in slots: _append_row( ev.id, "execute_session", f"平台兜底:时段{slot.get('time_slot', '')}执行会话(历史回填)", "platform", { "platform_supplemented": True, "time_slot": slot.get("time_slot"), "sessions": slot.get("sessions"), "completed": completed, "estimated": estimated, }, session, ) added += 1 else: _append_row( ev.id, "execute_session", "平台兜底:执行会话(历史回填)", "platform", {"platform_supplemented": True, "completed": completed, "estimated": estimated}, session, ) added += 1 if "start_analysis" not in types and sessions: _append_row( ev.id, "start_analysis", "平台兜底:所有会话已完成开始分析(历史回填)", "platform", {"platform_supplemented": True, "completed": completed, "estimated": estimated}, session, ) added += 1 return added def supplement_decision_logs(session: Session) -> int: """Platform audit backfill for decision logs. 方案③的决策日志由 OpenClaw agent 上报(LLM 自主,尽力而为)——异常路径 (如卡死恢复后重试)agent 可能跳过上报,导致决策过程页面为空。这里按评估 状态推导决策并补录: - EXECUTING:欠账(completed < estimated)补 execute_session,所有会话 完成后补 start_analysis。 - COMPLETED:历史评估/异常路径可能完全没有决策日志,回填 execute_session (按 plan 时段逐条)+ start_analysis,让旧报告也有决策过程可看。 只补"该类型缺失"的,不重复;且只记录状态,不改变 agent 的实际执行。 Returns: 补录的决策日志条数。 """ evals = session.exec( select(IntelligentEvalDB).where( IntelligentEvalDB.status.in_( [ IntelligentEvalStatus.EXECUTING.value, IntelligentEvalStatus.COMPLETED.value, ] ) ) ).all() added = 0 for ev in evals: plan = ev.get_plan() if ev.plan else {} estimated = plan.get("estimated_sessions", 0) sessions = session.exec( select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.eval_id == ev.id) ).all() completed = sum(1 for s in sessions if s.status == "completed") types = { x.decision_type for x in session.exec( select(IntelligentEvalDecisionLogDB).where(IntelligentEvalDecisionLogDB.eval_id == ev.id) ).all() } if ev.status == IntelligentEvalStatus.EXECUTING.value: added += _supplement_executing(ev, sessions, completed, estimated, types, session) elif ev.status == IntelligentEvalStatus.COMPLETED.value: added += _supplement_completed(ev, sessions, completed, estimated, plan, types, session) return added