Add the analysis role's execution path: a two-phase orchestration
(per-scenario diagnosis gathered in parallel, then a synthesis pass)
that reads the existing campaign report aggregation plus capped failure
samples, validates the LLM's JSON against the report schema, and strips
fabricated run/scenario references before persisting. Results upsert one
row per campaign (generating/completed/failed) with the model config
snapshot; GET/POST /api/campaigns/{id}/analysis expose the state machine,
guarding non-terminal campaigns and missing analysis models with 400s.
61 lines
1.6 KiB
Python
61 lines
1.6 KiB
Python
"""Shared test fixtures."""
|
|
|
|
import os
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
from sqlmodel import Session, SQLModel, create_engine
|
|
|
|
# Force tests to use a temp SQLite file instead of the real data/ dir.
|
|
os.environ.setdefault("AGENTEVAL_DATA_DIR_OVERRIDE", "")
|
|
|
|
|
|
@pytest.fixture()
|
|
def anyio_backend():
|
|
return "asyncio"
|
|
|
|
|
|
@pytest.fixture()
|
|
def tmp_db_path(tmp_path: Path) -> Path:
|
|
"""Return a path to a throwaway SQLite file for one test."""
|
|
return tmp_path / "test.db"
|
|
|
|
|
|
@pytest.fixture()
|
|
def db_session(tmp_db_path: Path) -> Session:
|
|
"""Yield a SQLModel Session backed by a fresh in-memory-ish SQLite file.
|
|
|
|
Tables are created via SQLModel.metadata.create_all; the session is closed
|
|
at the end of the test.
|
|
"""
|
|
# Import DB models so their table=True declarations register in metadata.
|
|
from agenteval.storage.db import ( # noqa: F401
|
|
CampaignAnalysisDB,
|
|
CampaignDB,
|
|
EvalResultDB,
|
|
EvalRunDB,
|
|
EvalTargetDB,
|
|
ModelConfigDB,
|
|
ScenarioDB,
|
|
ScenarioModelBindingDB,
|
|
TurnDB,
|
|
)
|
|
|
|
engine = create_engine(
|
|
f"sqlite:///{tmp_db_path}",
|
|
connect_args={"check_same_thread": False},
|
|
)
|
|
SQLModel.metadata.create_all(engine)
|
|
|
|
# Sanity check: verify the scenarios table has all expected columns.
|
|
from sqlalchemy import inspect as sa_inspect
|
|
cols = [c["name"] for c in sa_inspect(engine).get_columns("scenarios")]
|
|
assert "llm_config" in cols, f"scenarios table missing llm_config; cols={cols}"
|
|
|
|
session = Session(engine)
|
|
try:
|
|
yield session
|
|
finally:
|
|
session.close()
|
|
engine.dispose()
|