feat(report): compare requires same scenario version (ticket 05)
Some checks failed
CI / test (push) Failing after 39s

对比报告可比性收紧为同场景同考纲版本(ADR-0001):跨版本 API 返回 400
(detail 含双方版本号),报告生成层抛 ValueError;前端对比候选按
同场景 + 同版本过滤,A 变更后自动清空不可比的 B。文档"尚未实现"标注移除。
This commit is contained in:
sinohqb 2026-07-29 11:21:52 +08:00
parent 0a47260237
commit 770d260750
8 changed files with 80 additions and 14 deletions

View File

@ -4,10 +4,10 @@
**Blocked by:** 04 — 运行记录场景版本。 **Blocked by:** 04 — 运行记录场景版本。
**Status:** ready-for-agent **Status:** done
- [ ] 同场景不同版本的两次运行对比API 400detail 含双方版本号 - [x] 同场景不同版本的两次运行对比API 400detail 含双方版本号
- [ ] 同场景同版本对比正常生成(含动态用例场景——同考纲即可比) - [x] 同场景同版本对比正常生成(含动态用例场景——同考纲即可比)
- [ ] 报告生成层对跨版本对比抛出明确错误 - [x] 报告生成层对跨版本对比抛出明确错误
- [ ] 前端对比候选按同场景 + 同版本过滤;跨版本被拒时提示可读 - [x] 前端对比候选按同场景 + 同版本过滤;跨版本被拒时提示可读
- [ ] 集成与单元测试覆盖拒绝与放行两侧(先例:现有对比报告测试) - [x] 集成与单元测试覆盖拒绝与放行两侧(先例:现有对比报告测试)

View File

@ -41,7 +41,7 @@ _Avoid_: 测试点、题目
_Avoid_: 差异报告 _Avoid_: 差异报告
**场景版本Scenario Version**: **场景版本Scenario Version**:
场景考纲的版本标识。仅考纲字段用例集、模型绑定、LLM 配置)变更时递增;名称、描述、标签等元数据编辑不升版。(决策见 ADR-0001,尚未实现 场景考纲的版本标识。仅考纲字段用例集、模型绑定、LLM 配置)变更时递增;名称、描述、标签等元数据编辑不升版。(决策见 ADR-0001
_Avoid_: 修订号 _Avoid_: 修订号
**评测运行Run / EvalRun**: **评测运行Run / EvalRun**:

View File

@ -199,6 +199,12 @@ def generate_compare_report(run_id_1: str, run_id_2: str, session=None) -> dict[
# "changed" and the diff would be meaningless — reject early. # "changed" and the diff would be meaningless — reject early.
if report_a.get("scenario_id") != report_b.get("scenario_id"): if report_a.get("scenario_id") != report_b.get("scenario_id"):
raise ValueError("compare report requires both runs to use the same scenario") raise ValueError("compare report requires both runs to use the same scenario")
# 同场景还须同考纲版本才可比ADR-0001
if report_a.get("scenario_version") != report_b.get("scenario_version"):
raise ValueError(
"compare report requires the same scenario version "
f"(A: v{report_a.get('scenario_version')}, B: v{report_b.get('scenario_version')})"
)
def _summary_delta(key: str) -> float: def _summary_delta(key: str) -> float:
return report_b["summary"][key] - report_a["summary"][key] return report_b["summary"][key] - report_a["summary"][key]

View File

@ -31,6 +31,12 @@ def get_compare_report(
raise HTTPException(status_code=404, detail=f"run not found: {run2}") raise HTTPException(status_code=404, detail=f"run not found: {run2}")
if run_a.scenario_id != run_b.scenario_id: if run_a.scenario_id != run_b.scenario_id:
raise HTTPException(status_code=400, detail="对比报告要求两个运行使用相同场景") raise HTTPException(status_code=400, detail="对比报告要求两个运行使用相同场景")
if run_a.scenario_version != run_b.scenario_version:
raise HTTPException(
status_code=400,
detail=f"对比报告要求相同的场景考纲版本A: v{run_a.scenario_version}, B: v{run_b.scenario_version}"
"考纲变更前后的运行不可比",
)
return generate_compare_report(run1, run2, session) return generate_compare_report(run1, run2, session)

View File

@ -10,5 +10,5 @@
## Consequences ## Consequences
- 数据模型需加 scenario version 字段Run 需记录所用版本v0.5 实施项,尚未实现) - 数据模型已加 scenario version 字段Run 创建时快照所用版本v0.5 已实现)
- 动态用例每次运行题目不同不影响可比性——可比性单位是"同考纲"(同场景同版本),不是"同考卷" - 动态用例每次运行题目不同不影响可比性——可比性单位是"同考纲"(同场景同版本),不是"同考卷"

View File

@ -125,10 +125,11 @@ export default function ReportsPage() {
const handleView = (runId: string) => { const handleView = (runId: string) => {
setSearchParams({ run: runId }) setSearchParams({ run: runId })
loadReport(runId) loadReport(runId)
// 对比报告要求同场景A 变更后若 B 场景不同则清空 // 对比报告要求同场景同版本A 变更后若 B 不可比则清空
const a = runs.find((r) => r.id === runId) const a = runs.find((r) => r.id === runId)
const b = runs.find((r) => r.id === compareRunId) const b = runs.find((r) => r.id === compareRunId)
if (a && b && a.scenario_id !== b.scenario_id) setCompareRunId('') if (a && b && (a.scenario_id !== b.scenario_id
|| (a.scenario_version ?? 1) !== (b.scenario_version ?? 1))) setCompareRunId('')
} }
const handleCompare = async () => { const handleCompare = async () => {
@ -205,10 +206,14 @@ export default function ReportsPage() {
const runSelectOptions = filteredRuns.map(buildOption) const runSelectOptions = filteredRuns.map(buildOption)
// 报告 B 只能选与报告 A 同场景的 run // 报告 B 只能选与报告 A 同场景同版本(同考纲)的 run
const selectedRun = runs.find((r) => r.id === selectedRunId) const selectedRun = runs.find((r) => r.id === selectedRunId)
const compareOptions = runs const compareOptions = runs
.filter((r) => r.id !== selectedRunId && selectedRun && r.scenario_id === selectedRun.scenario_id) .filter((r) =>
r.id !== selectedRunId
&& selectedRun
&& r.scenario_id === selectedRun.scenario_id
&& (r.scenario_version ?? 1) === (selectedRun.scenario_version ?? 1))
.map(buildOption) .map(buildOption)
const optionFilter = (input: string, opt?: { searchText?: string }) => const optionFilter = (input: string, opt?: { searchText?: string }) =>
@ -291,7 +296,7 @@ export default function ReportsPage() {
filterOption={optionFilter} filterOption={optionFilter}
options={compareOptions} options={compareOptions}
popupMatchSelectWidth={false} popupMatchSelectWidth={false}
notFoundContent={<Empty image={Empty.PRESENTED_IMAGE_SIMPLE} description="没有同场景的其他评测记录" />} notFoundContent={<Empty image={Empty.PRESENTED_IMAGE_SIMPLE} description="没有同场景同版本的其他评测记录" />}
/> />
</Tooltip> </Tooltip>
<Button <Button

View File

@ -39,7 +39,12 @@ def client_with_db(tmp_path):
session.close() session.close()
def _seed_run(session: Session, name_suffix: str = "", scenario_id: str | None = None) -> str: def _seed_run(
session: Session,
name_suffix: str = "",
scenario_id: str | None = None,
scenario_version: int = 1,
) -> str:
target = EvalTarget( target = EvalTarget(
name=f"target{name_suffix}", name=f"target{name_suffix}",
platform=PlatformType.AI_DIGITAL_EMPLOYEE, platform=PlatformType.AI_DIGITAL_EMPLOYEE,
@ -60,6 +65,7 @@ def _seed_run(session: Session, name_suffix: str = "", scenario_id: str | None =
run = EvalRun( run = EvalRun(
target_id=target.id, target_id=target.id,
scenario_id=scenario_id, scenario_id=scenario_id,
scenario_version=scenario_version,
status=RunStatus.COMPLETED, status=RunStatus.COMPLETED,
) )
run = RunRepository(session).create(run) run = RunRepository(session).create(run)
@ -165,6 +171,29 @@ def test_compare_report_different_scenarios_400(client_with_db):
assert "相同场景" in resp.json()["detail"] assert "相同场景" in resp.json()["detail"]
def test_compare_report_cross_version_400(client_with_db):
"""同场景不同考纲版本 → 400提示含双方版本号ticket 05 / ADR-0001"""
client, session = client_with_db
run_id_a = _seed_run(session, "A")
sid = RunRepository(session).get(run_id_a).scenario_id
run_id_b = _seed_run(session, "B", scenario_id=sid, scenario_version=2)
resp = client.get(f"/api/reports/compare?run1={run_id_a}&run2={run_id_b}")
assert resp.status_code == 400
detail = resp.json()["detail"]
assert "v1" in detail and "v2" in detail
def test_compare_report_same_version_ok(client_with_db):
client, session = client_with_db
run_id_a = _seed_run(session, "A", scenario_version=3)
sid = RunRepository(session).get(run_id_a).scenario_id
run_id_b = _seed_run(session, "B", scenario_id=sid, scenario_version=3)
resp = client.get(f"/api/reports/compare?run1={run_id_a}&run2={run_id_b}")
assert resp.status_code == 200
assert resp.json()["run_a"]["scenario_version"] == 3
assert resp.json()["run_b"]["scenario_version"] == 3
def test_compare_report_run_not_found(client_with_db): def test_compare_report_run_not_found(client_with_db):
client, session = client_with_db client, session = client_with_db
run_id = _seed_run(session) run_id = _seed_run(session)

View File

@ -40,6 +40,7 @@ def _seed_run(
pass_rate: float = 1.0, pass_rate: float = 1.0,
n_cases: int = 1, n_cases: int = 1,
scenario_id: str | None = None, scenario_id: str | None = None,
scenario_version: int = 1,
connectivity_cases: int = 0, connectivity_cases: int = 0,
errored_cases: int = 0, errored_cases: int = 0,
) -> str: ) -> str:
@ -72,6 +73,7 @@ def _seed_run(
run = EvalRun( run = EvalRun(
target_id=target.id, target_id=target.id,
scenario_id=scenario_id, scenario_id=scenario_id,
scenario_version=scenario_version,
status=RunStatus.COMPLETED, status=RunStatus.COMPLETED,
) )
run = RunRepository(session).create(run) run = RunRepository(session).create(run)
@ -242,6 +244,24 @@ def test_compare_report_different_scenarios_rejected(report_session):
generate_compare_report(run_id_a, run_id_b, report_session) generate_compare_report(run_id_a, run_id_b, report_session)
def test_compare_report_cross_version_rejected(report_session):
"""同场景不同考纲版本不可比ticket 05 / ADR-0001"""
run_id_a = _seed_run(report_session, n_cases=1)
sid = _scenario_of(report_session, run_id_a)
run_id_b = _seed_run(report_session, n_cases=1, scenario_id=sid, scenario_version=2)
with pytest.raises(ValueError, match="version"):
generate_compare_report(run_id_a, run_id_b, report_session)
def test_compare_report_same_version_allowed(report_session):
run_id_a = _seed_run(report_session, n_cases=1, scenario_version=2)
sid = _scenario_of(report_session, run_id_a)
run_id_b = _seed_run(report_session, n_cases=1, scenario_id=sid, scenario_version=2)
result = generate_compare_report(run_id_a, run_id_b, report_session)
assert result["run_a"]["scenario_version"] == 2
assert result["run_b"]["scenario_version"] == 2
# ── connectivity case annotation (ticket 02) ───────────────────────────── # ── connectivity case annotation (ticket 02) ─────────────────────────────
def test_report_marks_connectivity_case(report_session): def test_report_marks_connectivity_case(report_session):