Files
Video-Tree-TRM5/tests/unit/test_run_store.py
T
iomgaa 6a5424a618 feat(question_gen): add SQLite run store for generation telemetry
Implements QuestionGenStore with:
- Idempotent schema initialization (question_gen_runs + question_gen_items)
- Run lifecycle: record_run_start / record_run_end / get_run_stats
- Per-item recording: record_item / update_gates / update_difficulty
- GateReportLike Protocol for duck-type gate report compatibility
- WAL mode + foreign keys + check_same_thread=False

DDL aligns with research-wiki/schemas/question-gen-{runs,items}.md.

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
2026-07-11 23:15:53 -04:00

245 lines
8.0 KiB
Python

"""QuestionGenStore 单元测试 — SQLite 出题管线日志记录器。
覆盖 schema 幂等性、run 生命周期、item 写入与门判定更新、难度更新。
"""
from __future__ import annotations
from dataclasses import dataclass
from enum import Enum
from typing import TYPE_CHECKING
import pytest
if TYPE_CHECKING:
from pathlib import Path
from app.question_gen.run_store import QuestionGenStore, RunStats
# ---------------------------------------------------------------------------
# GateReport 测试替身(Task 6 尚未实现,此处构造兼容 duck-type)
# ---------------------------------------------------------------------------
class _Verdict(Enum):
"""门判定枚举替身。"""
PASS = "pass"
FAIL = "fail"
SKIP = "skip"
@dataclass(frozen=True)
class _GateResult:
"""单门判定结果替身。"""
verdict: _Verdict
reason: str
@dataclass
class _MockGateReport:
"""GateReport 测试替身,duck-type 兼容。"""
key_verify: _GateResult
blind_answer: _GateResult
multi_true: _GateResult
leak_test: _GateResult
@property
def passed(self) -> bool:
"""所有门均 PASS 或 SKIP 时视为通过。"""
results = [self.key_verify, self.blind_answer, self.multi_true, self.leak_test]
return all(r.verdict != _Verdict.FAIL for r in results)
@property
def reject_reason(self) -> str | None:
"""返回首个 FAIL 门的 reason,若全部通过则 None。"""
for r in [self.key_verify, self.blind_answer, self.multi_true, self.leak_test]:
if r.verdict == _Verdict.FAIL:
return r.reason
return None
# ---------------------------------------------------------------------------
# Fixtures
# ---------------------------------------------------------------------------
@pytest.fixture()
def store(tmp_path: Path) -> QuestionGenStore:
"""创建临时 SQLite 上的 QuestionGenStore 实例。"""
db_path = tmp_path / "question_gen.db"
s = QuestionGenStore(db_path=db_path)
yield s
s.close()
# ---------------------------------------------------------------------------
# TestQuestionGenStore
# ---------------------------------------------------------------------------
class TestQuestionGenStore:
"""QuestionGenStore 全功能验证。"""
def test_schema_idempotent(self, tmp_path: Path) -> None:
"""连续两次初始化同一 db 不报错,表结构不变。"""
db_path = tmp_path / "idem.db"
s1 = QuestionGenStore(db_path=db_path)
s2 = QuestionGenStore(db_path=db_path)
# 第二次初始化不应抛异常
s2.record_run_start(run_id="r-idem", git_sha="abc123", config_snapshot="{}")
s1.close()
s2.close()
def test_record_run_lifecycle(self, store: QuestionGenStore) -> None:
"""start -> record items -> end -> get_stats 完整生命周期。"""
run_id = "run-001"
store.record_run_start(
run_id=run_id,
git_sha="deadbeef",
config_snapshot='{"n_slots": 240}',
)
# 写入几个 item 来模拟统计
for i in range(5):
store.record_item(
item_id=f"item-{i}",
run_id=run_id,
slot_id=f"slot-{i}",
video_id="v001",
family="retrieval",
task_type="Object Recognition",
skill_target="M1",
attempt=1,
question_text=f"问题{i}",
)
# 结束 run
stats = RunStats(total_slots=10, accepted=5, rejected=3, heavy_sampled=2)
store.record_run_end(run_id=run_id, status="completed", stats=stats)
# 查询 stats
retrieved = store.get_run_stats(run_id)
assert retrieved.total_slots == 10
assert retrieved.accepted == 5
assert retrieved.rejected == 3
assert retrieved.heavy_sampled == 2
def test_record_item_and_gates(self, store: QuestionGenStore) -> None:
"""record_item 后 update_gates 正确更新门判定列与 final_status。"""
run_id = "run-gates"
store.record_run_start(run_id=run_id, git_sha="aaa111", config_snapshot="{}")
item_id = "item-gate-1"
store.record_item(
item_id=item_id,
run_id=run_id,
slot_id="slot-0",
video_id="v002",
family="reasoning",
task_type="Action Sequence",
skill_target="M2",
attempt=1,
question_text="为什么这样做?",
)
# 所有门通过
report_pass = _MockGateReport(
key_verify=_GateResult(_Verdict.PASS, "ok"),
blind_answer=_GateResult(_Verdict.PASS, "ok"),
multi_true=_GateResult(_Verdict.SKIP, "n/a"),
leak_test=_GateResult(_Verdict.PASS, "ok"),
)
store.update_gates(item_id=item_id, report=report_pass)
# 验证数据库中的值
import sqlite3
conn = sqlite3.connect(str(store._db_path))
conn.row_factory = sqlite3.Row
row = conn.execute(
"SELECT * FROM question_gen_items WHERE item_id=?", (item_id,)
).fetchone()
conn.close()
assert row["gate_key_verify"] == "pass"
assert row["gate_blind_answer"] == "pass"
assert row["gate_multi_true"] == "skip"
assert row["gate_leak_test"] == "pass"
assert row["gate_reject_reason"] is None
assert row["final_status"] == "accepted"
def test_update_gates_rejected(self, store: QuestionGenStore) -> None:
"""门判定失败时 final_status 为 rejected 且记录 reject_reason。"""
run_id = "run-rej"
store.record_run_start(run_id=run_id, git_sha="bbb222", config_snapshot="{}")
item_id = "item-rej-1"
store.record_item(
item_id=item_id,
run_id=run_id,
slot_id="slot-1",
video_id="v003",
family="visual",
task_type="Scene Transition",
skill_target="M3",
attempt=2,
question_text="画面中有什么?",
)
report_fail = _MockGateReport(
key_verify=_GateResult(_Verdict.PASS, "ok"),
blind_answer=_GateResult(_Verdict.FAIL, "答案可由常识推断"),
multi_true=_GateResult(_Verdict.SKIP, "n/a"),
leak_test=_GateResult(_Verdict.SKIP, "n/a"),
)
store.update_gates(item_id=item_id, report=report_fail)
import sqlite3
conn = sqlite3.connect(str(store._db_path))
conn.row_factory = sqlite3.Row
row = conn.execute(
"SELECT * FROM question_gen_items WHERE item_id=?", (item_id,)
).fetchone()
conn.close()
assert row["final_status"] == "rejected"
assert row["gate_reject_reason"] == "答案可由常识推断"
def test_update_difficulty(self, store: QuestionGenStore) -> None:
"""update_difficulty 正确写入 difficulty_steps。"""
run_id = "run-diff"
store.record_run_start(run_id=run_id, git_sha="ccc333", config_snapshot="{}")
item_id = "item-diff-1"
store.record_item(
item_id=item_id,
run_id=run_id,
slot_id="slot-2",
video_id="v004",
family="enumeration",
task_type="Counting",
skill_target="M4",
attempt=1,
question_text="有多少个?",
)
store.update_difficulty(item_id=item_id, difficulty_steps=7)
import sqlite3
conn = sqlite3.connect(str(store._db_path))
conn.row_factory = sqlite3.Row
row = conn.execute(
"SELECT difficulty_steps FROM question_gen_items WHERE item_id=?",
(item_id,),
).fetchone()
conn.close()
assert row["difficulty_steps"] == 7
def test_get_run_stats_not_found(self, store: QuestionGenStore) -> None:
"""查询不存在的 run_id 应报错。"""
with pytest.raises(ValueError, match="run_id"):
store.get_run_stats("nonexistent-run")