Compare commits

..

3 Commits

Author SHA1 Message Date
sinohqb
5a81c570c0 style: fix ruff whitespace warnings
All checks were successful
CI / test (push) Successful in 3m16s
2026-08-24 01:53:35 +08:00
sinohqb
09ff2ed123 refactor(intelligent-eval): reduce nesting complexity in supplement_decision_logs
Extract helper functions _supplement_executing and _supplement_completed
to flatten the nested conditional logic. This improves readability and
makes the code easier to test and maintain.

Addresses code review finding: supplement_decision_logs nested complexity
2026-08-24 01:53:15 +08:00
sinohqb
da7dd434dd perf(intelligent-eval): 修复 N+1 查询和参数名混淆
- expire_stale_running_sessions: 使用单次 JOIN 查询替代 N+1 查询
  将每个会话单独查询最后消息时间改为一次性获取所有 running 会话及其最后消息时间

- submit_report/evals_needing_analyst_nudge: 消除 session/sessions 参数名混淆
  将局部变量 sessions 重命名为 eval_sessions,避免与数据库会话参数 session 混淆

这些改进提升了查询性能并增强了代码可读性。
2026-08-24 01:52:05 +08:00
2 changed files with 120 additions and 81 deletions

View File

@ -130,47 +130,16 @@ def list_decision_logs(eval_id: str, session: Session) -> list[dict]:
return [_log_to_dict(log) for log in logs]
def supplement_decision_logs(session: Session) -> int:
"""Platform audit backfill for decision logs.
方案③的决策日志由 OpenClaw agent 上报LLM 自主尽力而为异常路径
如卡死恢复后重试agent 可能跳过上报导致决策过程页面为空这里按评估
状态推导决策并补录
- EXECUTING欠账completed < estimated execute_session所有会话
完成后补 start_analysis
- COMPLETED历史评估/异常路径可能完全没有决策日志回填 execute_session
plan 时段逐条+ start_analysis让旧报告也有决策过程可看
只补"该类型缺失"不重复且只记录状态不改变 agent 的实际执行
Returns:
补录的决策日志条数
"""
evals = session.exec(
select(IntelligentEvalDB).where(
IntelligentEvalDB.status.in_(
[
IntelligentEvalStatus.EXECUTING.value,
IntelligentEvalStatus.COMPLETED.value,
]
)
)
).all()
def _supplement_executing(
ev: IntelligentEvalDB,
sessions: list,
completed: int,
estimated: int,
types: set[str],
session: Session,
) -> int:
"""Supplement decision logs for EXECUTING evals."""
added = 0
for ev in evals:
plan = ev.get_plan() if ev.plan else {}
estimated = plan.get("estimated_sessions", 0)
sessions = session.exec(select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.eval_id == ev.id)).all()
completed = sum(1 for s in sessions if s.status == "completed")
types = {
x.decision_type
for x in session.exec(
select(IntelligentEvalDecisionLogDB).where(IntelligentEvalDecisionLogDB.eval_id == ev.id)
).all()
}
if ev.status == IntelligentEvalStatus.EXECUTING.value:
if "execute_session" not in types and completed < estimated:
_append_row(
ev.id,
@ -191,8 +160,20 @@ def supplement_decision_logs(session: Session) -> int:
session,
)
added += 1
elif ev.status == IntelligentEvalStatus.COMPLETED.value:
# 历史回填completed 评估决策日志全缺失时,按时段补 execute_session
return added
def _supplement_completed(
ev: IntelligentEvalDB,
sessions: list,
completed: int,
estimated: int,
plan: dict,
types: set[str],
session: Session,
) -> int:
"""Supplement decision logs for COMPLETED evals (historical backfill)."""
added = 0
if "execute_session" not in types:
slots = plan.get("time_distribution") or []
if slots:
@ -233,3 +214,54 @@ def supplement_decision_logs(session: Session) -> int:
)
added += 1
return added
def supplement_decision_logs(session: Session) -> int:
"""Platform audit backfill for decision logs.
方案③的决策日志由 OpenClaw agent 上报LLM 自主尽力而为异常路径
如卡死恢复后重试agent 可能跳过上报导致决策过程页面为空这里按评估
状态推导决策并补录
- EXECUTING欠账completed < estimated execute_session所有会话
完成后补 start_analysis
- COMPLETED历史评估/异常路径可能完全没有决策日志回填 execute_session
plan 时段逐条+ start_analysis让旧报告也有决策过程可看
只补"该类型缺失"不重复且只记录状态不改变 agent 的实际执行
Returns:
补录的决策日志条数
"""
evals = session.exec(
select(IntelligentEvalDB).where(
IntelligentEvalDB.status.in_(
[
IntelligentEvalStatus.EXECUTING.value,
IntelligentEvalStatus.COMPLETED.value,
]
)
)
).all()
added = 0
for ev in evals:
plan = ev.get_plan() if ev.plan else {}
estimated = plan.get("estimated_sessions", 0)
sessions = session.exec(
select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.eval_id == ev.id)
).all()
completed = sum(1 for s in sessions if s.status == "completed")
types = {
x.decision_type
for x in session.exec(
select(IntelligentEvalDecisionLogDB).where(IntelligentEvalDecisionLogDB.eval_id == ev.id)
).all()
}
if ev.status == IntelligentEvalStatus.EXECUTING.value:
added += _supplement_executing(ev, sessions, completed, estimated, types, session)
elif ev.status == IntelligentEvalStatus.COMPLETED.value:
added += _supplement_completed(ev, sessions, completed, estimated, plan, types, session)
return added

View File

@ -226,8 +226,8 @@ def submit_report(session: Session, eval_id: str, report: dict[str, Any]) -> Int
不完整证据不再阻塞报告提交
"""
repo = IntelligentEvalRepository(session)
sessions = IntelligentEvalSessionRepository(session).list_by_eval(eval_id)
if any(s.status == IntelligentEvalSessionStatus.RUNNING for s in sessions):
eval_sessions = IntelligentEvalSessionRepository(session).list_by_eval(eval_id)
if any(s.status == IntelligentEvalSessionStatus.RUNNING for s in eval_sessions):
raise IntelligentEvalTransitionError("存在进行中的会话,不能提交报告")
# ADR-0011submit 边界把 scores 归一到 {overall, dimensions} 单一规范结构
if report.get("scores"):
@ -432,29 +432,36 @@ def expire_stale_running_sessions(session: Session) -> int:
now = utc_now()
# SQLite 读出为 naive datetime阈值须同为 naive 才能在 Python 侧比较
threshold = now.replace(tzinfo=None) - timedelta(minutes=SESSION_IDLE_EXPIRE_MINUTES)
rows = session.exec(
select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.status == "running")
).all()
# 单次查询获取所有 running 会话及其最后消息时间(避免 N+1 查询)
stmt = (
select(
IntelligentEvalSessionDB,
func.max(IntelligentEvalMessageDB.created_at).label("last_message_at"),
)
.outerjoin(
IntelligentEvalMessageDB,
IntelligentEvalMessageDB.session_id == IntelligentEvalSessionDB.id,
)
.where(IntelligentEvalSessionDB.status == "running")
.group_by(IntelligentEvalSessionDB.id)
)
rows = session.exec(stmt).all()
expired = 0
for row in rows:
last_message_at = session.exec(
select(func.max(IntelligentEvalMessageDB.created_at)).where(
IntelligentEvalMessageDB.session_id == row.id
)
).one()
last_activity = last_message_at or row.created_at
for session_row, last_message_at in rows:
last_activity = last_message_at or session_row.created_at
if last_activity is None or last_activity >= threshold:
continue
row.status = IntelligentEvalSessionStatus.EXPIRED.value
row.closed_at = now
session_row.status = IntelligentEvalSessionStatus.EXPIRED.value
session_row.closed_at = now
expired += 1
append_decision_log(
row.eval_id,
session_row.eval_id,
"session_expired",
f"平台兜底:会话 {SESSION_IDLE_EXPIRE_MINUTES} 分钟无新轮次,置为过期(不完整证据)",
"platform",
{"platform_supplemented": True, "session_id": row.id, "turn_count": row.turn_count},
{"platform_supplemented": True, "session_id": session_row.id, "turn_count": session_row.turn_count},
session,
)
@ -674,16 +681,16 @@ def evals_needing_analyst_nudge(session: Session) -> list[str]:
needing: list[str] = []
for row in executing:
sessions = session.exec(
eval_sessions = session.exec(
select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.eval_id == row.id)
).all()
if not sessions or any(s.status not in _TERMINAL_SESSION_STATUSES for s in sessions):
if not eval_sessions or any(s.status not in _TERMINAL_SESSION_STATUSES for s in eval_sessions):
continue
# 冒烟教训:窗口未结束且会话数未达计划时,未来时段到期后还要建会话,
# 此时催促 analyst 会让报告提前收敛(漏掉后续时段的证据)
if _window_has_pending_future_slots(row, sessions, now):
if _window_has_pending_future_slots(row, eval_sessions, now):
continue
closed_moments = [s.closed_at for s in sessions if s.closed_at is not None]
closed_moments = [s.closed_at for s in eval_sessions if s.closed_at is not None]
if not closed_moments:
continue
last_closed = max(closed_moments)