AgentEvalTool/tests/integration/test_worker_skill_api.py
sinohqb eb4944a8bd feat(intelligent-eval): terminal-state discipline watchdogs (ADR-0011)
常见故障自愈有上限,超限收敛终态且可见:任务 attempts 上限、会话过期、
planning 双闸、executing 超窗兜底、触发失败计数判死、孤儿 agent 双管、
fire-and-forget 触发;open_session 预算硬闸门、settle 按终态区分、报告
scores 归一化;cron 池遗留面全删。
2026-08-20 14:34:17 +08:00

146 lines
4.3 KiB
Python

"""Integration tests for worker skill APIs (decision logs)."""
import pytest
from agenteval.intelligent_eval.models import IntelligentEvalStatus
from agenteval.storage.db import (
IntelligentEvalDB,
IntelligentEvalDecisionLogDB,
)
from agenteval.web.app import app
from agenteval.web.deps import get_db
from fastapi.testclient import TestClient
from sqlmodel import Session, SQLModel, create_engine, select
@pytest.fixture()
def client(tmp_path):
"""Create a TestClient with a fresh database."""
from agenteval.storage.db import ( # noqa: F401
IntelligentEvalDB,
IntelligentEvalDecisionLogDB,
)
engine = create_engine(
f"sqlite:///{tmp_path / 'test.db'}",
connect_args={"check_same_thread": False},
)
SQLModel.metadata.create_all(engine)
session = Session(engine)
def override_get_db():
try:
yield session
finally:
pass
app.dependency_overrides[get_db] = override_get_db
client = TestClient(app)
yield client
app.dependency_overrides.clear()
session.close()
engine.dispose()
@pytest.fixture()
def db_session(client):
"""Get the database session from the client fixture."""
return next(app.dependency_overrides[get_db]())
def test_create_decision_log(client: TestClient, db_session: Session):
"""Test creating a decision log."""
# Create eval
eval_db = IntelligentEvalDB(
name="test",
target_id="target1",
status=IntelligentEvalStatus.EXECUTING.value,
)
db_session.add(eval_db)
db_session.commit()
# Create decision log
response = client.post(
f"/api/intelligent-evals/{eval_db.id}/decision-logs",
json={
"decision_type": "execute_session",
"reason": "时段 8-10h 欠账 2 个会话",
"context": {
"current_slot": "8-10h",
"deficit": 2,
"completed_sessions": 1,
},
"cron_id": "cron-123",
},
)
assert response.status_code == 200
data = response.json()
assert data["eval_id"] == eval_db.id
assert data["decision_type"] == "execute_session"
assert data["reason"] == "时段 8-10h 欠账 2 个会话"
assert data["context"]["current_slot"] == "8-10h"
assert data["cron_id"] == "cron-123"
# Verify log saved to DB
log = db_session.exec(
select(IntelligentEvalDecisionLogDB).where(IntelligentEvalDecisionLogDB.eval_id == eval_db.id)
).first()
assert log is not None
assert log.decision_type == "execute_session"
def test_create_decision_log_eval_not_found(client: TestClient):
"""Test creating decision log for non-existent eval."""
response = client.post(
"/api/intelligent-evals/nonexistent/decision-logs",
json={
"decision_type": "wait",
"reason": "test",
"context": {},
"cron_id": "cron-123",
},
)
assert response.status_code == 404
def test_decision_log_multiple_entries(client: TestClient, db_session: Session):
"""Test creating multiple decision logs for same eval."""
# Create eval
eval_db = IntelligentEvalDB(
name="test",
target_id="target1",
status=IntelligentEvalStatus.EXECUTING.value,
)
db_session.add(eval_db)
db_session.commit()
# Create 3 decision logs
decisions = [
("execute_session", "时段到期"),
("wait", "当前时段无欠账"),
("start_analysis", "所有会话完成"),
]
for decision_type, reason in decisions:
response = client.post(
f"/api/intelligent-evals/{eval_db.id}/decision-logs",
json={
"decision_type": decision_type,
"reason": reason,
"context": {},
"cron_id": "cron-123",
},
)
assert response.status_code == 200
# Verify all logs saved
logs = db_session.exec(
select(IntelligentEvalDecisionLogDB)
.where(IntelligentEvalDecisionLogDB.eval_id == eval_db.id)
.order_by(IntelligentEvalDecisionLogDB.created_at)
).all()
assert len(logs) == 3
assert logs[0].decision_type == "execute_session"
assert logs[1].decision_type == "wait"
assert logs[2].decision_type == "start_analysis"