From 71543f042ae283a64951422622d2740df4ff2c14 Mon Sep 17 00:00:00 2001 From: sinohqb Date: Mon, 24 Aug 2026 05:47:00 +0800 Subject: [PATCH] =?UTF-8?q?refactor(intelligent-eval):=20=E5=8F=AF?= =?UTF-8?q?=E8=A7=81=E6=80=A7=E6=8E=A5=E7=BC=9D=E6=94=B6=E6=95=9B=EF=BC=88?= =?UTF-8?q?Phase=201=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 将「已删即 404」语义收进 IntelligentEvalRepository 单一接缝,消除三处独立裁决; 任务监控开始隐藏已删评估的任务(本 Phase 唯一刻意行为变化)。 - repository.py 新增 visible() 谓词与 require_live_eval() 服务接缝; 六处裸谓词统一走它,get()/get_including_deleted() 语义不变。 - decision_logs.py 删除本地 _require_eval,三处调用迁至 repository 接缝; count_decisions 由 len(.all()) 改为 func.count。 - task_queue.py list_tasks 与 stats 过滤已删评估的任务(行为变化)。 - web/routers/intelligent_evals.py: _require_eval_exists → _require_live_eval, 把 LookupError 翻译为 404;expired 会话 Markdown 标注下沉至 read_model.report_markdown_by_eval;配置快照 11 字段序列化收至 config_snapshot.snapshot_to_dict 单一出口。 - AGENTS.md 登记可见性纪律(已知陷阱 #6)。 - 补 characterization 测试锁定四处契约;更新 task_queue 测试以使用 真实 eval_id(可见性过滤后字面 eval_id 不再可见)。 --- AGENTS.md | 2 + .../intelligent_eval/config_snapshot.py | 17 ++ .../intelligent_eval/decision_logs.py | 35 ++- .../agenteval/intelligent_eval/read_model.py | 17 ++ .../agenteval/intelligent_eval/repository.py | 46 ++-- .../agenteval/intelligent_eval/task_queue.py | 18 +- .../web/routers/intelligent_evals.py | 75 ++----- tests/conftest.py | 1 + tests/integration/conftest.py | 8 + .../test_intelligent_eval_task_queue_api.py | 5 +- ...ligent_eval_visibility_characterization.py | 204 ++++++++++++++++++ 11 files changed, 320 insertions(+), 108 deletions(-) create mode 100644 tests/integration/conftest.py create mode 100644 tests/integration/test_intelligent_eval_visibility_characterization.py diff --git a/AGENTS.md b/AGENTS.md index 6ef6a53..a8245fa 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -197,3 +197,5 @@ tests/ 4. **SQLite 连接池**:必须用 `StaticPool` + `check_same_thread=False`。使用默认连接池会导致多线程下连接耗尽(async + FastAPI 线程池混合场景)。 5. **`.env` 在 t480 上需手动维护**:部署脚本不会自动同步 `.env` 文件,新增环境变量需 SSH 到主机手动补全。容易导致 OpenClaw 代理等功能静默失效。 + +6. **智能评估可见性接缝**:逻辑删除(`deleted` 状态)语义只住在 `intelligent_eval/repository.py`——查询走 `IntelligentEvalRepository.visible()` 谓词,存在性校验走 `require_live_eval`(路由层翻译为 404)。禁止在其他模块直查 `IntelligentEvalDB` 判断删除态,否则删除语义会在多处漂移(Phase 1 收敛前的教训:快照、决策日志、任务监控各自实现了一遍"已删即 404")。 diff --git a/backend/agenteval/intelligent_eval/config_snapshot.py b/backend/agenteval/intelligent_eval/config_snapshot.py index 4c48815..1533dc4 100644 --- a/backend/agenteval/intelligent_eval/config_snapshot.py +++ b/backend/agenteval/intelligent_eval/config_snapshot.py @@ -70,6 +70,23 @@ def get_snapshot(snapshot_id: str, session: Session) -> Optional[IntelligentEval return session.get(IntelligentEvalConfigSnapshotDB, snapshot_id) +def snapshot_to_dict(s: IntelligentEvalConfigSnapshotDB) -> dict[str, Any]: + """快照的稳定序列化形状(列表与详情共用的单一出口)。""" + return { + "id": s.id, + "eval_id": s.eval_id, + "snapshot_type": s.snapshot_type, + "goal": s.goal, + "seeds": s.get_seeds(), + "intent": s.intent, + "role_description": s.role_description, + "time_window_hours": s.time_window_hours, + "plan": s.get_plan(), + "created_at": s.created_at.isoformat() if s.created_at else None, + "created_by": s.created_by, + } + + def compare_snapshots( snapshot1: IntelligentEvalConfigSnapshotDB, snapshot2: IntelligentEvalConfigSnapshotDB, diff --git a/backend/agenteval/intelligent_eval/decision_logs.py b/backend/agenteval/intelligent_eval/decision_logs.py index 94821e0..e3d41c2 100644 --- a/backend/agenteval/intelligent_eval/decision_logs.py +++ b/backend/agenteval/intelligent_eval/decision_logs.py @@ -3,11 +3,14 @@ Pulled out of ``web/routers/intelligent_evals.py`` so the router only handles HTTP validation and error translation. The ORM writes and reads now live here. """ + from typing import Any +from sqlalchemy import func from sqlmodel import Session, select from agenteval.intelligent_eval.models import IntelligentEvalStatus +from agenteval.intelligent_eval.repository import IntelligentEvalRepository from agenteval.storage.db import ( IntelligentEvalDB, IntelligentEvalDecisionLogDB, @@ -27,12 +30,6 @@ def _log_to_dict(log: IntelligentEvalDecisionLogDB) -> dict: } -def _require_eval(eval_id: str, session: Session) -> None: - eval_db = session.get(IntelligentEvalDB, eval_id) - if eval_db is None or eval_db.status == IntelligentEvalStatus.DELETED.value: - raise LookupError(f"intelligent eval {eval_id} not found") - - def _append_row( eval_id: str, decision_type: str, @@ -71,7 +68,7 @@ def create_decision_log( Raises ``LookupError`` if eval not found. """ - _require_eval(eval_id, session) + IntelligentEvalRepository(session).require_live_eval(eval_id) # Dedupe: same (eval, decision_type, context) → return existing. # Limit to 100 records to avoid loading too many into memory; in practice, @@ -105,25 +102,25 @@ def append_decision_log( Raises ``LookupError`` if eval not found. """ - _require_eval(eval_id, session) + IntelligentEvalRepository(session).require_live_eval(eval_id) return _append_row(eval_id, decision_type, reason, cron_id, context, session) def count_decisions(eval_id: str, decision_type: str, session: Session) -> int: """该评估某类型决策日志的条数(平台闸门计数与 attempt 落账的计数原语)。""" - return len( - session.exec( - select(IntelligentEvalDecisionLogDB).where( - IntelligentEvalDecisionLogDB.eval_id == eval_id, - IntelligentEvalDecisionLogDB.decision_type == decision_type, - ) - ).all() - ) + return session.exec( + select(func.count()) + .select_from(IntelligentEvalDecisionLogDB) + .where( + IntelligentEvalDecisionLogDB.eval_id == eval_id, + IntelligentEvalDecisionLogDB.decision_type == decision_type, + ) + ).one() def list_decision_logs(eval_id: str, session: Session) -> list[dict]: """List decision logs for an eval. Raises ``LookupError`` if eval not found.""" - _require_eval(eval_id, session) + IntelligentEvalRepository(session).require_live_eval(eval_id) logs = session.exec( select(IntelligentEvalDecisionLogDB) @@ -252,9 +249,7 @@ def supplement_decision_logs(session: Session) -> int: for ev in evals: plan = ev.get_plan() if ev.plan else {} estimated = plan.get("estimated_sessions", 0) - sessions = session.exec( - select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.eval_id == ev.id) - ).all() + sessions = session.exec(select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.eval_id == ev.id)).all() completed = sum(1 for s in sessions if s.status == "completed") types = { x.decision_type diff --git a/backend/agenteval/intelligent_eval/read_model.py b/backend/agenteval/intelligent_eval/read_model.py index 534b51a..f355c5b 100644 --- a/backend/agenteval/intelligent_eval/read_model.py +++ b/backend/agenteval/intelligent_eval/read_model.py @@ -22,6 +22,7 @@ from agenteval.intelligent_eval.models import ( IntelligentEvalSessionStatus, IntelligentEvalStatus, ) +from agenteval.intelligent_eval.report import render_report_markdown from agenteval.intelligent_eval.repository import ( IntelligentEvalMessageRepository, IntelligentEvalRepository, @@ -345,3 +346,19 @@ class IntelligentEvalReadModel: if eval_obj is None or eval_obj.report is None: return None return eval_obj.name, eval_obj.report + + def report_markdown_by_eval(self, eval_id: str) -> str | None: + """报告的 Markdown 投影(含 expired 会话的不完整证据标注,ADR-0011)。""" + + report = self.report_by_eval(eval_id) + if report is None: + return None + name, payload = report + markdown = render_report_markdown(payload, name=name, eval_id=eval_id) + expired = [s for s in self._sessions.list_by_eval(eval_id) if s.status is IntelligentEvalSessionStatus.EXPIRED] + if expired: + lines = ["", "## 不完整证据会话", ""] + for s in expired: + lines.append(f"- 会话 `{s.id}`({s.turn_count} 轮):60 分钟无新轮次过期,证据不完整") + markdown = markdown.rstrip() + "\n" + "\n".join(lines) + "\n" + return markdown diff --git a/backend/agenteval/intelligent_eval/repository.py b/backend/agenteval/intelligent_eval/repository.py index 184b23d..b8928f2 100644 --- a/backend/agenteval/intelligent_eval/repository.py +++ b/backend/agenteval/intelligent_eval/repository.py @@ -97,26 +97,25 @@ class IntelligentEvalRepository: completed_at=db.completed_at, ) + @staticmethod + def visible(): + """可见性谓词:逻辑删除的评估默认不可见(查询位点共用接缝)。""" + return IntelligentEvalDB.status != IntelligentEvalStatus.DELETED.value + + @staticmethod + def _is_visible(db: IntelligentEvalDB) -> bool: + return db.status != IntelligentEvalStatus.DELETED.value + def list_all(self) -> list[IntelligentEval]: - statement = ( - select(IntelligentEvalDB) - .where(IntelligentEvalDB.status != IntelligentEvalStatus.DELETED.value) - .order_by(IntelligentEvalDB.created_at.desc()) - ) + statement = select(IntelligentEvalDB).where(self.visible()).order_by(IntelligentEvalDB.created_at.desc()) return [self._from_db(r) for r in self.session.exec(statement).all()] - def list_page( - self, offset: int, limit: int, status: Optional[str] = None - ) -> list[IntelligentEval]: + def list_page(self, offset: int, limit: int, status: Optional[str] = None) -> list[IntelligentEval]: """Return one page of evals (created_at desc), optionally filtered by status. Deleted evals are always excluded — the list API has no recycle bin. """ - statement = ( - select(IntelligentEvalDB) - .where(IntelligentEvalDB.status != IntelligentEvalStatus.DELETED.value) - .order_by(IntelligentEvalDB.created_at.desc()) - ) + statement = select(IntelligentEvalDB).where(self.visible()).order_by(IntelligentEvalDB.created_at.desc()) if status: statement = statement.where(IntelligentEvalDB.status == status) statement = statement.offset(offset).limit(limit) @@ -124,17 +123,13 @@ class IntelligentEvalRepository: def count(self) -> int: """Total number of evaluations (for pagination metadata).""" - return self.session.exec( - select(func.count()) - .select_from(IntelligentEvalDB) - .where(IntelligentEvalDB.status != IntelligentEvalStatus.DELETED.value) - ).one() + return self.session.exec(select(func.count()).select_from(IntelligentEvalDB).where(self.visible())).one() def count_by_status(self) -> dict[str, int]: """Count evaluations per status (for the list page stat bar).""" rows = self.session.exec( select(IntelligentEvalDB.status, func.count(IntelligentEvalDB.id)) - .where(IntelligentEvalDB.status != IntelligentEvalStatus.DELETED.value) + .where(self.visible()) .group_by(IntelligentEvalDB.status) ).all() stats = {status.value: 0 for status in IntelligentEvalStatus if status != IntelligentEvalStatus.DELETED} @@ -144,10 +139,17 @@ class IntelligentEvalRepository: def get(self, eval_id: str) -> Optional[IntelligentEval]: db = self.session.get(IntelligentEvalDB, eval_id) - if db is None or db.status == IntelligentEvalStatus.DELETED.value: + if db is None or not self._is_visible(db): return None return self._from_db(db) + def require_live_eval(self, eval_id: str) -> IntelligentEval: + """可见性单一接缝:存在且未删 → 返回实体,否则 ``LookupError``。""" + ev = self.get(eval_id) + if ev is None: + raise LookupError(f"intelligent eval {eval_id} not found") + return ev + def get_including_deleted(self, eval_id: str) -> Optional[IntelligentEval]: """包含已删除评估的原样读取(供 delete 幂等检查使用)。""" db = self.session.get(IntelligentEvalDB, eval_id) @@ -168,7 +170,7 @@ class IntelligentEvalRepository: ) .where( IntelligentEvalDB.id == eval_id, - IntelligentEvalDB.status != IntelligentEvalStatus.DELETED.value, + self.visible(), ) .order_by(IntelligentEvalSessionDB.created_at.asc()) ) @@ -285,6 +287,7 @@ class IntelligentEvalRepository: }, ) + class IntelligentEvalSessionRepository: """CRUD for intelligent eval sessions.""" @@ -443,6 +446,7 @@ class IntelligentEvalSessionRepository: return CompareAndSetStatus.NOT_FOUND, None return CompareAndSetStatus.APPLIED, self._from_db(db) + class IntelligentEvalMessageRepository: """CRUD for intelligent eval session messages.""" diff --git a/backend/agenteval/intelligent_eval/task_queue.py b/backend/agenteval/intelligent_eval/task_queue.py index 3b834fd..e495743 100644 --- a/backend/agenteval/intelligent_eval/task_queue.py +++ b/backend/agenteval/intelligent_eval/task_queue.py @@ -13,6 +13,7 @@ from typing import Optional from sqlmodel import Session, func, select, update from agenteval.intelligent_eval.models import IntelligentEvalStatus +from agenteval.intelligent_eval.repository import IntelligentEvalRepository from agenteval.storage.db import ( IntelligentEvalDB, IntelligentEvalTaskQueueDB, @@ -259,9 +260,7 @@ def settle_tasks_for_finished_evals(session: Session) -> int: from agenteval.storage.db import IntelligentEvalTaskQueueDB tasks = session.exec( - select(IntelligentEvalTaskQueueDB).where( - IntelligentEvalTaskQueueDB.status.in_(["pending", "assigned"]) - ) + select(IntelligentEvalTaskQueueDB).where(IntelligentEvalTaskQueueDB.status.in_(["pending", "assigned"])) ).all() settled = 0 @@ -305,7 +304,14 @@ def list_tasks( Returns: {"tasks": [...], "stats": {pending, assigned, completed, failed, unresolved}} """ - stmt = select(IntelligentEvalTaskQueueDB).order_by(IntelligentEvalTaskQueueDB.created_at.desc()) + # 可见性接缝:已删评估的任务不出现在监控列表与统计中 + live_eval_ids = select(IntelligentEvalDB.id).where(IntelligentEvalRepository.visible()) + + stmt = ( + select(IntelligentEvalTaskQueueDB) + .where(IntelligentEvalTaskQueueDB.eval_id.in_(live_eval_ids)) + .order_by(IntelligentEvalTaskQueueDB.created_at.desc()) + ) if status: stmt = stmt.where(IntelligentEvalTaskQueueDB.status == status) if eval_id: @@ -318,7 +324,9 @@ def list_tasks( select( IntelligentEvalTaskQueueDB.status, func.count(IntelligentEvalTaskQueueDB.id), - ).group_by(IntelligentEvalTaskQueueDB.status) + ) + .where(IntelligentEvalTaskQueueDB.eval_id.in_(live_eval_ids)) + .group_by(IntelligentEvalTaskQueueDB.status) ).all() for status_val, cnt in rows: if status_val in stats: diff --git a/backend/agenteval/web/routers/intelligent_evals.py b/backend/agenteval/web/routers/intelligent_evals.py index aa89626..aaf2a89 100644 --- a/backend/agenteval/web/routers/intelligent_evals.py +++ b/backend/agenteval/web/routers/intelligent_evals.py @@ -17,10 +17,9 @@ from agenteval.intelligent_eval.lifecycle import ( IntelligentEvalNotFoundError, IntelligentEvalTransitionError, ) -from agenteval.intelligent_eval.models import IntelligentEvalSessionStatus from agenteval.intelligent_eval.read_model import IntelligentEvalReadModel -from agenteval.intelligent_eval.report import ReportModel, render_report_markdown -from agenteval.intelligent_eval.repository import IntelligentEvalSessionRepository +from agenteval.intelligent_eval.report import ReportModel +from agenteval.intelligent_eval.repository import IntelligentEvalRepository from agenteval.web.deps import get_db router = APIRouter() @@ -70,14 +69,12 @@ def _translate(exc: Exception) -> HTTPException: return HTTPException(status_code=409, detail=exc.reason) -def _require_eval_exists(session: Session, eval_id: str) -> None: - """404 when the eval is missing or logically deleted.""" - from agenteval.intelligent_eval.models import IntelligentEvalStatus - from agenteval.storage.db import IntelligentEvalDB - - eval_db = session.get(IntelligentEvalDB, eval_id) - if eval_db is None or eval_db.status == IntelligentEvalStatus.DELETED.value: - raise HTTPException(status_code=404, detail=f"intelligent eval {eval_id} not found") +def _require_live_eval(session: Session, eval_id: str) -> None: + """把可见性接缝翻译成 HTTP:缺失或已删 → 404。""" + try: + IntelligentEvalRepository(session).require_live_eval(eval_id) + except LookupError as exc: + raise HTTPException(status_code=404, detail=str(exc)) from exc def _eval_response(ev, session: Session) -> dict: @@ -234,22 +231,9 @@ async def get_report(eval_id: str, session: Session = Depends(get_db)) -> dict: @router.get("/{eval_id}/report/markdown", response_class=PlainTextResponse) async def get_report_markdown(eval_id: str, session: Session = Depends(get_db)) -> PlainTextResponse: - report = IntelligentEvalReadModel(session).report_by_eval(eval_id) - if report is None: + markdown = IntelligentEvalReadModel(session).report_markdown_by_eval(eval_id) + if markdown is None: raise HTTPException(status_code=404, detail="report not submitted yet") - name, payload = report - markdown = render_report_markdown(payload, name=name, eval_id=eval_id) - # ADR-0011:expired 会话按不完整证据标注,提醒读者结论证据不全 - expired_sessions = [ - s - for s in IntelligentEvalSessionRepository(session).list_by_eval(eval_id) - if s.status == IntelligentEvalSessionStatus.EXPIRED - ] - if expired_sessions: - lines = ["", "## 不完整证据会话", ""] - for s in expired_sessions: - lines.append(f"- 会话 `{s.id}`({s.turn_count} 轮):60 分钟无新轮次过期,证据不完整") - markdown = markdown.rstrip() + "\n" + "\n".join(lines) + "\n" return PlainTextResponse(markdown, media_type="text/markdown; charset=utf-8") @@ -393,27 +377,10 @@ async def list_config_snapshots(eval_id: str, session: Session = Depends(get_db) """List all config snapshots for an evaluation.""" from agenteval.intelligent_eval import config_snapshot - _require_eval_exists(session, eval_id) + _require_live_eval(session, eval_id) snapshots = config_snapshot.list_snapshots(eval_id, session) - return { - "snapshots": [ - { - "id": s.id, - "eval_id": s.eval_id, - "snapshot_type": s.snapshot_type, - "goal": s.goal, - "seeds": s.get_seeds(), - "intent": s.intent, - "role_description": s.role_description, - "time_window_hours": s.time_window_hours, - "plan": s.get_plan(), - "created_at": s.created_at.isoformat() if s.created_at else None, - "created_by": s.created_by, - } - for s in snapshots - ] - } + return {"snapshots": [config_snapshot.snapshot_to_dict(s) for s in snapshots]} @router.get("/{eval_id}/config-snapshots/{snapshot_id}") @@ -421,25 +388,13 @@ async def get_config_snapshot(eval_id: str, snapshot_id: str, session: Session = """Get a single config snapshot.""" from agenteval.intelligent_eval import config_snapshot - _require_eval_exists(session, eval_id) + _require_live_eval(session, eval_id) snapshot = config_snapshot.get_snapshot(snapshot_id, session) if snapshot is None or snapshot.eval_id != eval_id: raise HTTPException(status_code=404, detail=f"snapshot {snapshot_id} not found") - return { - "id": snapshot.id, - "eval_id": snapshot.eval_id, - "snapshot_type": snapshot.snapshot_type, - "goal": snapshot.goal, - "seeds": snapshot.get_seeds(), - "intent": snapshot.intent, - "role_description": snapshot.role_description, - "time_window_hours": snapshot.time_window_hours, - "plan": snapshot.get_plan(), - "created_at": snapshot.created_at.isoformat() if snapshot.created_at else None, - "created_by": snapshot.created_by, - } + return config_snapshot.snapshot_to_dict(snapshot) class CompareSnapshotsRequest(BaseModel): @@ -456,7 +411,7 @@ async def compare_config_snapshots( """Compare two config snapshots and return differences.""" from agenteval.intelligent_eval import config_snapshot - _require_eval_exists(session, eval_id) + _require_live_eval(session, eval_id) # Get both snapshots snapshot1 = config_snapshot.get_snapshot(request.snapshot_id_1, session) diff --git a/tests/conftest.py b/tests/conftest.py index 0ec8c9b..0b9fe2f 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -37,6 +37,7 @@ def db_session(tmp_db_path: Path) -> Session: EvalTargetDB, ExplorationMessageDB, ExplorationSessionDB, + IntelligentEvalConfigSnapshotDB, IntelligentEvalDB, IntelligentEvalDecisionLogDB, IntelligentEvalMessageDB, diff --git a/tests/integration/conftest.py b/tests/integration/conftest.py new file mode 100644 index 0000000..c974d73 --- /dev/null +++ b/tests/integration/conftest.py @@ -0,0 +1,8 @@ +"""Integration-level shared fixtures. + +Re-exports the in-process app fixtures from ``test_intelligent_evals_api`` +so newer integration test modules can consume them without importing +fixture objects into test-module namespaces (which ruff flags as F811). +""" + +from .test_intelligent_evals_api import client, seeded_db # noqa: F401 diff --git a/tests/integration/test_intelligent_eval_task_queue_api.py b/tests/integration/test_intelligent_eval_task_queue_api.py index 136b920..3fc978c 100644 --- a/tests/integration/test_intelligent_eval_task_queue_api.py +++ b/tests/integration/test_intelligent_eval_task_queue_api.py @@ -257,8 +257,9 @@ def test_list_tasks(client: TestClient, db_session: Session): def test_list_tasks_status_filter(client: TestClient, db_session: Session): """Status filter narrows the task list.""" - db_session.add(IntelligentEvalTaskQueueDB(eval_id="eval1", status="pending", priority=1, reason="slot_due")) - db_session.add(IntelligentEvalTaskQueueDB(eval_id="eval1", status="failed", priority=1, reason="slot_due")) + eval_db = _make_eval(db_session, "status-filter-eval", days_ago=0) + db_session.add(IntelligentEvalTaskQueueDB(eval_id=eval_db.id, status="pending", priority=1, reason="slot_due")) + db_session.add(IntelligentEvalTaskQueueDB(eval_id=eval_db.id, status="failed", priority=1, reason="slot_due")) db_session.commit() response = client.get("/api/intelligent-evals/tasks?status=failed") diff --git a/tests/integration/test_intelligent_eval_visibility_characterization.py b/tests/integration/test_intelligent_eval_visibility_characterization.py new file mode 100644 index 0000000..6a64023 --- /dev/null +++ b/tests/integration/test_intelligent_eval_visibility_characterization.py @@ -0,0 +1,204 @@ +"""Characterization tests: intelligent-eval visibility semantics (Phase 1). + +锁定逻辑删除可见性收敛(require_live_eval 接缝)后的行为契约。覆盖四处: + +1. 「已删即 404」的统一裁决:config-snapshots 详情/对比端点与 + decision_logs 服务层都经 require_live_eval 接缝。 +2. 任务队列监控**隐藏**已删评估的任务(Phase 1.6 唯一的刻意行为变化, + 锁定变更后的新契约)。 +3. config-snapshots 端点的响应形状(11 字段序列化,snapshot_to_dict 单一出口)。 +4. expired 会话在 Markdown 报告中的「不完整证据」标注文本(ADR-0011)。 +""" + +import pytest +from agenteval.intelligent_eval.models import IntelligentEvalStatus +from agenteval.storage.db import ( + IntelligentEvalSessionDB, + IntelligentEvalTaskQueueDB, + utc_now, +) +from sqlmodel import select + +from .test_intelligent_evals_api import ( + _create_eval, + _create_executing_eval, + _create_session, + _report_payload, + _submit_plan, +) + + +async def _complete_eval(client, eval_id: str) -> None: + resp = await client.put(f"/api/intelligent-evals/{eval_id}/report", json={"report": _report_payload()}) + assert resp.status_code == 200, resp.text + + +class TestSnapshotEndpointDeletedSemantics: + """路由 _require_eval_exists 路径:config-snapshots 详情与对比。""" + + async def test_deleted_eval_snapshot_detail_returns_404(self, client, seeded_db): + eval_id = await _create_executing_eval(client) + await _complete_eval(client, eval_id) + snapshots = (await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots")).json()["snapshots"] + assert snapshots + await client.delete(f"/api/intelligent-evals/{eval_id}") + + resp = await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots/{snapshots[0]['id']}") + assert resp.status_code == 404 + + async def test_deleted_eval_snapshot_compare_returns_404(self, client, seeded_db): + eval_id = (await _create_eval(client))["id"] + await _submit_plan(client, eval_id) # second snapshot (plan_submitted) + await client.post(f"/api/intelligent-evals/{eval_id}/approve") + await _complete_eval(client, eval_id) + snapshots = (await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots")).json()["snapshots"] + assert len(snapshots) >= 2 + await client.delete(f"/api/intelligent-evals/{eval_id}") + + resp = await client.post( + f"/api/intelligent-evals/{eval_id}/config-snapshots/compare", + json={"snapshot_id_1": snapshots[0]["id"], "snapshot_id_2": snapshots[1]["id"]}, + ) + assert resp.status_code == 404 + + +class TestDecisionLogServiceDeletedSemantics: + """decision_logs 经 require_live_eval 接缝:已删评估一律 LookupError。""" + + async def test_deleted_eval_service_calls_raise_lookup_error(self, client, seeded_db): + from agenteval.intelligent_eval import decision_logs + + eval_id = await _create_executing_eval(client) + await _complete_eval(client, eval_id) + await client.delete(f"/api/intelligent-evals/{eval_id}") + + with pytest.raises(LookupError): + decision_logs.list_decision_logs(eval_id, seeded_db) + with pytest.raises(LookupError): + decision_logs.create_decision_log(eval_id, "wait", "late report", "w-1", {}, seeded_db) + with pytest.raises(LookupError): + decision_logs.append_decision_log(eval_id, "wait", "late bookkeeping", "platform", {}, seeded_db) + + async def test_repo_get_hides_deleted_but_including_deleted_returns_it(self, client, seeded_db): + from agenteval.intelligent_eval.repository import IntelligentEvalRepository + + eval_id = await _create_executing_eval(client) + await _complete_eval(client, eval_id) + await client.delete(f"/api/intelligent-evals/{eval_id}") + + repo = IntelligentEvalRepository(seeded_db) + assert repo.get(eval_id) is None + hidden = repo.get_including_deleted(eval_id) + assert hidden is not None + assert hidden.status is IntelligentEvalStatus.DELETED + + +class TestTaskMonitorHidesDeletedEvals: + """Phase 1.6 刻意行为变化:任务监控不再泄漏已删评估的任务。""" + + async def test_list_tasks_hides_tasks_of_deleted_eval(self, client, seeded_db): + eval_id = await _create_executing_eval(client) + seeded_db.add( + IntelligentEvalTaskQueueDB( + eval_id=eval_id, + status="completed", + priority=1, + reason="slot_due", + completed_at=utc_now(), + ) + ) + seeded_db.commit() + + await _complete_eval(client, eval_id) + await client.delete(f"/api/intelligent-evals/{eval_id}") + + resp = await client.get("/api/intelligent-evals/tasks") + assert resp.status_code == 200 + data = resp.json() + + assert [t for t in data["tasks"] if t["eval_id"] == eval_id] == [] + assert data["stats"]["completed"] == 0 + + by_eval = await client.get("/api/intelligent-evals/tasks", params={"eval_id": eval_id}) + assert by_eval.json()["tasks"] == [] + + +class TestSnapshotResponseShape: + """锁定 config-snapshots 列表/详情的 11 字段序列化形状。""" + + SNAPSHOT_KEYS = { + "id", + "eval_id", + "snapshot_type", + "goal", + "seeds", + "intent", + "role_description", + "time_window_hours", + "plan", + "created_at", + "created_by", + } + + async def test_list_shape_and_snapshot_types(self, client, seeded_db): + eval_id = (await _create_eval(client))["id"] + await _submit_plan(client, eval_id) + + snapshots = (await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots")).json()["snapshots"] + assert len(snapshots) == 2 + assert {s["snapshot_type"] for s in snapshots} == {"created", "plan_submitted"} + for s in snapshots: + assert set(s.keys()) == self.SNAPSHOT_KEYS + assert s["eval_id"] == eval_id + created = next(s for s in snapshots if s["snapshot_type"] == "created") + assert created["created_by"] == "user" + assert created["goal"] == "评估退货流程处理能力" + assert created["time_window_hours"] == 24 + assert created["plan"] is None + + async def test_detail_shape_matches_list_item(self, client, seeded_db): + eval_id = (await _create_eval(client))["id"] + listed = (await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots")).json()["snapshots"][0] + + detail = (await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots/{listed['id']}")).json() + assert set(detail.keys()) == self.SNAPSHOT_KEYS + assert detail == listed + + async def test_snapshot_of_other_eval_returns_404(self, client, seeded_db): + eval_id = (await _create_eval(client))["id"] + other_id = (await _create_eval(client, name="另一个评估"))["id"] + foreign = (await client.get(f"/api/intelligent-evals/{other_id}/config-snapshots")).json()["snapshots"][0] + + resp = await client.get(f"/api/intelligent-evals/{eval_id}/config-snapshots/{foreign['id']}") + assert resp.status_code == 404 + + +class TestExpiredSessionMarkdownAnnotation: + """ADR-0011:expired 会话在 Markdown 报告尾部标注不完整证据。""" + + async def test_markdown_annotates_expired_sessions(self, client, seeded_db): + eval_id = await _create_executing_eval(client) + session = await _create_session(client, eval_id) + + row = seeded_db.exec(select(IntelligentEvalSessionDB).where(IntelligentEvalSessionDB.id == session["id"])).one() + row.status = "expired" + row.turn_count = 3 + row.closed_at = utc_now() + seeded_db.add(row) + seeded_db.commit() + + await _complete_eval(client, eval_id) + + resp = await client.get(f"/api/intelligent-evals/{eval_id}/report/markdown") + assert resp.status_code == 200 + text = resp.text + assert "## 不完整证据会话" in text + assert f"- 会话 `{session['id']}`(3 轮):60 分钟无新轮次过期,证据不完整" in text + + async def test_markdown_without_expired_sessions_has_no_annotation(self, client, seeded_db): + eval_id = await _create_executing_eval(client) + await _complete_eval(client, eval_id) + + resp = await client.get(f"/api/intelligent-evals/{eval_id}/report/markdown") + assert resp.status_code == 200 + assert "不完整证据会话" not in resp.text