diff --git a/app/question_gen/gates.py b/app/question_gen/gates.py new file mode 100644 index 0000000..39bce50 --- /dev/null +++ b/app/question_gen/gates.py @@ -0,0 +1,471 @@ +"""轻量四门质量检查 — 对生成题目执行 4 个独立的 LLM 质量门控。 + +四门: +1. key_verify: 验证答案在来源素材中有证据支撑。 +2. blind_answer: 无上下文时 LLM 能否答对(若答对 → 题目泄漏)。 +3. multi_true: 检测是否有多个选项可被视为正确。 +4. leak_test: 按家族特定模板探测答题捷径。 + +设计要点: +- run_gates 先做 verbatim_ratio 前置短路(> 0.5 直接 FAIL key_verify)。 +- 四门并发执行(asyncio.gather)。 +- 每门加载 store/prompts/question_gen/ 下对应模板,构造 messages,调 LLM。 +- LLM 返回 JSON {"verdict": "pass"|"fail", "reason": "..."},解析失败视为 FAIL。 +- session_id 必须透传至每次 LLM 调用(遥测关联)。 +""" + +from __future__ import annotations + +import asyncio +import json +from dataclasses import dataclass, field +from enum import Enum +from pathlib import Path +from typing import TYPE_CHECKING + +from loguru import logger + +if TYPE_CHECKING: + from app.question_gen.families import QuestionFamilySpec + from app.question_gen.postprocess import PostprocessResult + from app.tree.index import TreeIndex + from core.protocols import LLMProvider + +# --------------------------------------------------------------------------- +# 常量 +# --------------------------------------------------------------------------- + +_PROMPTS_DIR = Path(__file__).resolve().parent.parent.parent / "store" / "prompts" / "question_gen" + +_VERBATIM_THRESHOLD = 0.5 + + +# --------------------------------------------------------------------------- +# CandidateQuestion(Task 5 尚未实现,本地定义,后续迁移至 generator_v2.py) +# --------------------------------------------------------------------------- + + +@dataclass(frozen=True) +class CandidateQuestion: + """候选题目 — 出题管线生成、待门控审核的题目数据。 + + 属性: + question_id: 题目唯一标识。 + video_id: 所属视频标识。 + task_type: 题型(如 "Action Reasoning")。 + skill_target: 目标失败机制编号(M1-M5)。 + question: 题目文本。 + options: 选项元组(如 ("A. ...", "B. ...", "C. ...", "D. ..."))。 + answer: 正确答案字母(如 "A")。 + source_nodes: 来源节点 ID 元组。 + difficulty: 难度等级。 + subtitle_sentences: 字幕句子元组。 + frame_paths: 帧图片路径元组。 + """ + + question_id: str + video_id: str + task_type: str + skill_target: str + question: str + options: tuple[str, ...] + answer: str + source_nodes: tuple[str, ...] + difficulty: str + subtitle_sentences: tuple[str, ...] = field(default_factory=tuple) + frame_paths: tuple[str, ...] = field(default_factory=tuple) + + +# --------------------------------------------------------------------------- +# 门控结果类型 +# --------------------------------------------------------------------------- + + +class GateVerdict(Enum): + """门控判决枚举。""" + + PASS = "pass" + FAIL = "fail" + SKIP = "skip" + + +@dataclass(frozen=True) +class GateResult: + """单门判决结果。 + + 属性: + verdict: 门控判决(PASS / FAIL / SKIP)。 + reason: 判决原因说明。 + raw_response: LLM 原始返回内容。 + """ + + verdict: GateVerdict + reason: str + raw_response: str + + +@dataclass(frozen=True) +class GateReport: + """四门判决汇总报告。 + + 属性: + key_verify: 关键验证门结果。 + blind_answer: 盲答门结果。 + multi_true: 多正确门结果。 + leak_test: 泄漏测试门结果。 + """ + + key_verify: GateResult + blind_answer: GateResult + multi_true: GateResult + leak_test: GateResult + + @property + def passed(self) -> bool: + """全门 PASS 或 SKIP 时视为通过。""" + for gate in (self.key_verify, self.blind_answer, self.multi_true, self.leak_test): + if gate.verdict == GateVerdict.FAIL: + return False + return True + + @property + def reject_reason(self) -> str | None: + """首个 FAIL 门的 reason,全部通过返回 None。""" + for name, gate in [ + ("key_verify", self.key_verify), + ("blind_answer", self.blind_answer), + ("multi_true", self.multi_true), + ("leak_test", self.leak_test), + ]: + if gate.verdict == GateVerdict.FAIL: + return f"[{name}] {gate.reason}" + return None + + +# --------------------------------------------------------------------------- +# 内部工具函数 +# --------------------------------------------------------------------------- + + +def _load_prompt_template(filename: str) -> str: + """加载 store/prompts/question_gen/ 下的模板文件。 + + 参数: + filename: 模板文件名(如 "gate_key_verify.md")。 + + 返回: + 模板内容字符串。 + + 异常: + FileNotFoundError: 模板文件不存在。 + """ + path = _PROMPTS_DIR / filename + if not path.exists(): + msg = f"门控模板文件不存在: {path}" + raise FileNotFoundError(msg) + return path.read_text(encoding="utf-8") + + +def _resolve_source_text(candidate: CandidateQuestion, tree: TreeIndex) -> str: + """从树中解析候选题的来源节点文本。 + + 遍历树的所有层级,匹配 source_nodes 中的 ID,提取对应描述文本。 + + 参数: + candidate: 候选题目。 + tree: 视频树索引。 + + 返回: + 拼接的来源节点描述文本。 + """ + texts: list[str] = [] + target_ids = set(candidate.source_nodes) + + for l1 in tree.roots: + if l1.id in target_ids: + texts.append(f"[L1 {l1.id}] {l1.card.scene_summary}") + for l2 in l1.children: + if l2.id in target_ids: + texts.append(f"[L2 {l2.id}] {l2.card.event_description}") + for l3 in l2.children: + if l3.id in target_ids: + texts.append(f"[L3 {l3.id}] {l3.card.frame_summary}") + + if not texts: + logger.warning( + "未找到来源节点: candidate={}, source_nodes={}", + candidate.question_id, + candidate.source_nodes, + ) + return "(no source material found)" + + return "\n".join(texts) + + +def _format_options(options: tuple[str, ...]) -> str: + """将选项元组格式化为可读字符串。""" + return "\n".join(options) + + +def _parse_gate_response(raw_content: str) -> tuple[GateVerdict, str]: + """解析 LLM 返回的门控 JSON 响应。 + + 期望格式: {"verdict": "pass"|"fail", "reason": "..."} + 解析失败时返回 FAIL + parse_error。 + + 参数: + raw_content: LLM 返回的原始文本。 + + 返回: + (verdict, reason) 元组。 + """ + try: + # 尝试从文本中提取 JSON(可能被 markdown 代码块包裹) + content = raw_content.strip() + if "```" in content: + # 提取代码块中的内容 + parts = content.split("```") + for part in parts: + stripped = part.strip() + if stripped.startswith("json"): + stripped = stripped[4:].strip() + if stripped.startswith("{"): + content = stripped + break + + data = json.loads(content) + verdict_str = data.get("verdict", "").lower() + reason = data.get("reason", "") + + if verdict_str == "pass": + return GateVerdict.PASS, reason + elif verdict_str == "fail": + return GateVerdict.FAIL, reason + else: + return GateVerdict.FAIL, f"parse_error: invalid verdict '{verdict_str}'" + + except (json.JSONDecodeError, AttributeError, TypeError) as e: + return GateVerdict.FAIL, f"parse_error: {e}" + + +# --------------------------------------------------------------------------- +# 四门实现 +# --------------------------------------------------------------------------- + + +async def _gate_key_verify( + candidate: CandidateQuestion, + tree: TreeIndex, + llm: LLMProvider, + *, + session_id: str, +) -> GateResult: + """关键验证门 — 检查答案在来源素材中是否有证据支撑。 + + 参数: + candidate: 候选题目。 + tree: 视频树索引。 + llm: LLM 调用端口。 + session_id: 会话 ID(遥测关联)。 + + 返回: + GateResult 实例。 + """ + source_text = _resolve_source_text(candidate, tree) + template = _load_prompt_template("gate_key_verify.md") + prompt = template.format( + source_text=source_text, + question=candidate.question, + options=_format_options(candidate.options), + answer=candidate.answer, + ) + + response = await llm.chat( + [{"role": "user", "content": prompt}], + session_id=session_id, + ) + + verdict, reason = _parse_gate_response(response.content) + return GateResult(verdict=verdict, reason=reason, raw_response=response.content) + + +async def _gate_blind_answer( + candidate: CandidateQuestion, + llm: LLMProvider, + *, + session_id: str, +) -> GateResult: + """盲答门 — 无上下文时 LLM 能否答对(答对 → 题目泄漏)。 + + 参数: + candidate: 候选题目。 + llm: LLM 调用端口。 + session_id: 会话 ID(遥测关联)。 + + 返回: + GateResult 实例。 + """ + template = _load_prompt_template("gate_blind_answer.md") + prompt = template.format( + question=candidate.question, + options=_format_options(candidate.options), + ) + + response = await llm.chat( + [{"role": "user", "content": prompt}], + session_id=session_id, + ) + + verdict, reason = _parse_gate_response(response.content) + return GateResult(verdict=verdict, reason=reason, raw_response=response.content) + + +async def _gate_multi_true( + candidate: CandidateQuestion, + tree: TreeIndex, + llm: LLMProvider, + *, + session_id: str, +) -> GateResult: + """多正确门 — 检测是否有多个选项可被视为正确。 + + 参数: + candidate: 候选题目。 + tree: 视频树索引。 + llm: LLM 调用端口。 + session_id: 会话 ID(遥测关联)。 + + 返回: + GateResult 实例。 + """ + source_text = _resolve_source_text(candidate, tree) + template = _load_prompt_template("gate_multi_true.md") + prompt = template.format( + source_text=source_text, + question=candidate.question, + options=_format_options(candidate.options), + ) + + response = await llm.chat( + [{"role": "user", "content": prompt}], + session_id=session_id, + ) + + verdict, reason = _parse_gate_response(response.content) + return GateResult(verdict=verdict, reason=reason, raw_response=response.content) + + +async def _gate_leak_test( + candidate: CandidateQuestion, + family_spec: QuestionFamilySpec, + llm: LLMProvider, + *, + session_id: str, +) -> GateResult: + """泄漏测试门 — 按家族特定模板探测答题捷径。 + + 参数: + candidate: 候选题目。 + family_spec: 问题家族规格(含 leak_profile)。 + llm: LLM 调用端口。 + session_id: 会话 ID(遥测关联)。 + + 返回: + GateResult 实例。 + """ + probe_template_name = family_spec.leak_profile.probe_template + template = _load_prompt_template(probe_template_name) + prompt = template.format( + question=candidate.question, + options=_format_options(candidate.options), + answer=candidate.answer, + ) + + response = await llm.chat( + [{"role": "user", "content": prompt}], + session_id=session_id, + ) + + verdict, reason = _parse_gate_response(response.content) + return GateResult(verdict=verdict, reason=reason, raw_response=response.content) + + +# --------------------------------------------------------------------------- +# 编排入口 +# --------------------------------------------------------------------------- + + +async def run_gates( + candidate: CandidateQuestion, + tree: TreeIndex, + llm: LLMProvider, + family_spec: QuestionFamilySpec, + postprocess: PostprocessResult, + *, + session_id: str, +) -> GateReport: + """编排四门并发执行,返回汇总报告。 + + 前置规则:verbatim_ratio > 0.5 时直接短路 key_verify 为 FAIL(不调 LLM)。 + + 参数: + candidate: 候选题目。 + tree: 视频树索引。 + llm: LLM 调用端口。 + family_spec: 问题家族规格。 + postprocess: 后处理结果(含 verbatim_ratio)。 + session_id: 会话 ID(遥测关联)。 + + 返回: + GateReport 四门汇总。 + """ + # Phase 1: verbatim 前置短路 + if postprocess.verbatim_ratio > _VERBATIM_THRESHOLD: + logger.info( + "verbatim_ratio={:.3f} > {:.1f},key_verify 短路 FAIL", + postprocess.verbatim_ratio, + _VERBATIM_THRESHOLD, + ) + key_verify_result = GateResult( + verdict=GateVerdict.FAIL, + reason=f"verbatim_ratio={postprocess.verbatim_ratio:.3f} exceeds threshold {_VERBATIM_THRESHOLD}", + raw_response="", + ) + # 其余三门仍然并发执行(收集完整诊断信息) + blind_result, multi_result, leak_result = await asyncio.gather( + _gate_blind_answer(candidate, llm, session_id=session_id), + _gate_multi_true(candidate, tree, llm, session_id=session_id), + _gate_leak_test(candidate, family_spec, llm, session_id=session_id), + ) + return GateReport( + key_verify=key_verify_result, + blind_answer=blind_result, + multi_true=multi_result, + leak_test=leak_result, + ) + + # Phase 2: 四门并发执行 + key_result, blind_result, multi_result, leak_result = await asyncio.gather( + _gate_key_verify(candidate, tree, llm, session_id=session_id), + _gate_blind_answer(candidate, llm, session_id=session_id), + _gate_multi_true(candidate, tree, llm, session_id=session_id), + _gate_leak_test(candidate, family_spec, llm, session_id=session_id), + ) + + report = GateReport( + key_verify=key_result, + blind_answer=blind_result, + multi_true=multi_result, + leak_test=leak_result, + ) + + if report.passed: + logger.debug("四门全部通过: question_id={}", candidate.question_id) + else: + logger.info( + "门控拒绝: question_id={}, reason={}", + candidate.question_id, + report.reject_reason, + ) + + return report diff --git a/store/prompts/question_gen/gate_blind_answer.md b/store/prompts/question_gen/gate_blind_answer.md new file mode 100644 index 0000000..d3ca5f8 --- /dev/null +++ b/store/prompts/question_gen/gate_blind_answer.md @@ -0,0 +1,24 @@ +# Blind Answer Gate + +You are evaluating whether a multiple-choice question can be answered correctly **without any video context**, using only common sense or option patterns. + +## Question + +{question} + +## Options + +{options} + +## Instructions + +1. Read the question and options WITHOUT any source material. +2. Try to determine the correct answer using only common sense, option length patterns, or grammatical cues. +3. If you can confidently pick the correct answer → verdict "fail" (the question leaks information). +4. If you cannot determine the answer without context → verdict "pass" (the question genuinely requires video understanding). + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_key_verify.md b/store/prompts/question_gen/gate_key_verify.md new file mode 100644 index 0000000..57ac54a --- /dev/null +++ b/store/prompts/question_gen/gate_key_verify.md @@ -0,0 +1,35 @@ +# Key Verify Gate + +You are a quality-control judge for video understanding questions. + +## Task + +Given the source material from a video and a multiple-choice question with its designated correct answer, determine whether the correct answer is **supported by evidence** in the source material. + +## Source Material + +{source_text} + +## Question + +{question} + +## Options + +{options} + +## Designated Correct Answer + +{answer} + +## Instructions + +1. Read the source material carefully. +2. Determine if the designated correct answer can be derived or inferred from the source material. +3. If evidence supports the answer, verdict is "pass". If not, verdict is "fail". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_leak_enumeration.md b/store/prompts/question_gen/gate_leak_enumeration.md new file mode 100644 index 0000000..09534cc --- /dev/null +++ b/store/prompts/question_gen/gate_leak_enumeration.md @@ -0,0 +1,32 @@ +# Leak Test: Option Length Shortcut (ENUMERATION Family) + +You are detecting whether a question exploits option length patterns as a shortcut. + +## Question + +{question} + +## Options + +{options} + +## Designated Correct Answer + +{answer} + +## Instructions + +Check if the question can be answered by exploiting the length or specificity of options, without actually understanding the video content. + +1. Is the correct answer notably longer or more specific than distractors? +2. Are distractor options clearly shorter, vaguer, or less detailed than the correct answer? +3. Could a student "game" this question by always picking the longest/most-detailed option? + +If option length is an exploitable shortcut → verdict "fail". +If options are roughly balanced in length and specificity → verdict "pass". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_leak_reasoning.md b/store/prompts/question_gen/gate_leak_reasoning.md new file mode 100644 index 0000000..0b230ea --- /dev/null +++ b/store/prompts/question_gen/gate_leak_reasoning.md @@ -0,0 +1,32 @@ +# Leak Test: Frequency Shortcut (REASONING Family) + +You are detecting whether a question exploits word frequency patterns as a shortcut. + +## Question + +{question} + +## Options + +{options} + +## Designated Correct Answer + +{answer} + +## Instructions + +Check if the question can be answered by picking the option that shares the most words or phrases with the question stem, without genuine causal/logical reasoning. + +1. Does the correct answer have significantly more word overlap with the question than distractors? +2. Are distractor options phrased in notably different vocabulary from the question? +3. Could a student "game" this question by matching keywords between question and options? + +If frequency-based word matching is an exploitable shortcut → verdict "fail". +If the question requires genuine reasoning → verdict "pass". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_leak_retrieval.md b/store/prompts/question_gen/gate_leak_retrieval.md new file mode 100644 index 0000000..caa8cb3 --- /dev/null +++ b/store/prompts/question_gen/gate_leak_retrieval.md @@ -0,0 +1,32 @@ +# Leak Test: Temporal Proximity Shortcut (RETRIEVAL Family) + +You are detecting whether a question exploits temporal proximity patterns as a shortcut. + +## Question + +{question} + +## Options + +{options} + +## Designated Correct Answer + +{answer} + +## Instructions + +Check if the question can be answered by simply picking the event that is temporally closest to the question's time reference, without truly understanding the content. + +1. Does the correct answer correspond to the most recently mentioned or chronologically nearest event? +2. Are distractor options clearly from distant time points, making elimination trivial? +3. Could a student "game" this question by always picking the temporally proximate option? + +If temporal proximity is an exploitable shortcut → verdict "fail". +If the question requires genuine content understanding → verdict "pass". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_leak_spatial.md b/store/prompts/question_gen/gate_leak_spatial.md new file mode 100644 index 0000000..4bbd69e --- /dev/null +++ b/store/prompts/question_gen/gate_leak_spatial.md @@ -0,0 +1,32 @@ +# Leak Test: Spatial Default Shortcut (SPATIAL Family) + +You are detecting whether a question exploits spatial default assumptions as a shortcut. + +## Question + +{question} + +## Options + +{options} + +## Designated Correct Answer + +{answer} + +## Instructions + +Check if the question can be answered by relying on common spatial assumptions (e.g., "center of frame", "left to right", "foreground") without actual visual understanding. + +1. Is the correct answer the "default" spatial position people would assume (e.g., center, front)? +2. Are distractor options in positions that seem intuitively unlikely (e.g., extreme edges, behind)? +3. Could a student "game" this question by always picking the spatially default/expected option? + +If spatial defaults are an exploitable shortcut → verdict "fail". +If the question requires genuine spatial reasoning → verdict "pass". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_leak_visual.md b/store/prompts/question_gen/gate_leak_visual.md new file mode 100644 index 0000000..a342dec --- /dev/null +++ b/store/prompts/question_gen/gate_leak_visual.md @@ -0,0 +1,32 @@ +# Leak Test: Visual Salience Shortcut (VISUAL Family) + +You are detecting whether a question exploits visual salience patterns as a shortcut. + +## Question + +{question} + +## Options + +{options} + +## Designated Correct Answer + +{answer} + +## Instructions + +Check if the question can be answered by simply picking the most visually salient or "obvious" object/action, without careful visual analysis. + +1. Is the correct answer the most prominent or commonly expected element in such a scene? +2. Are distractor options obviously implausible or uncommon objects/actions for the described setting? +3. Could a student "game" this question by guessing the most visually dominant element? + +If visual salience is an exploitable shortcut → verdict "fail". +If the question requires careful visual discrimination → verdict "pass". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation"}} +``` diff --git a/store/prompts/question_gen/gate_multi_true.md b/store/prompts/question_gen/gate_multi_true.md new file mode 100644 index 0000000..e020e60 --- /dev/null +++ b/store/prompts/question_gen/gate_multi_true.md @@ -0,0 +1,28 @@ +# Multi-True Gate + +You are checking whether a multiple-choice question has **more than one plausibly correct answer** given the source material. + +## Source Material + +{source_text} + +## Question + +{question} + +## Options + +{options} + +## Instructions + +1. Read the source material and the question carefully. +2. For each option, assess whether it could be considered a correct or plausible answer given the source. +3. If exactly ONE option is clearly correct → verdict "pass". +4. If TWO or more options are plausibly correct → verdict "fail". + +## Response Format (strict JSON) + +```json +{{"verdict": "pass" or "fail", "reason": "brief explanation listing plausible options if multiple"}} +``` diff --git a/tests/unit/test_gates.py b/tests/unit/test_gates.py new file mode 100644 index 0000000..4329ca6 --- /dev/null +++ b/tests/unit/test_gates.py @@ -0,0 +1,312 @@ +"""轻量四门质量检查单元测试。""" + +from __future__ import annotations + +import json +from typing import Any + +import pytest + +from app.question_gen.families import RETRIEVAL_FAMILY +from app.question_gen.gates import ( + CandidateQuestion, + GateVerdict, + _gate_blind_answer, + _gate_key_verify, + _gate_leak_test, + _gate_multi_true, + run_gates, +) +from app.question_gen.postprocess import PostprocessResult +from app.tree.index import ( + IndexMeta, + L1Card, + L1Node, + L2Card, + L2Node, + L3Card, + L3Node, + TreeIndex, +) +from core.types import LLMResponse + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + + +def _make_llm_response(verdict: str, reason: str) -> LLMResponse: + """构造一个 LLM 返回值,content 为标准 JSON 格式。""" + return LLMResponse( + content=json.dumps({"verdict": verdict, "reason": reason}), + thinking="", + model="mock-model", + provider="mock", + prompt_tokens=10, + completion_tokens=5, + latency_ms=100, + ttft_ms=None, + max_inter_token_ms=None, + cache_hit=False, + call_id="mock-call-001", + ) + + +class MockLLM: + """可配置的 LLM mock — 按调用顺序返回预设响应。""" + + def __init__(self, responses: list[LLMResponse]) -> None: + self._responses = list(responses) + self._call_count = 0 + self.calls: list[dict[str, Any]] = [] + + async def chat( + self, + messages: list[dict[str, Any]], + *, + session_id: str | None = None, + parent_call_id: str | None = None, + ) -> LLMResponse: + """记录调用并返回预设响应。""" + self.calls.append({"messages": messages, "session_id": session_id}) + idx = self._call_count + self._call_count += 1 + if idx < len(self._responses): + return self._responses[idx] + # 默认返回 PASS + return _make_llm_response("pass", "default") + + +@pytest.fixture() +def mock_llm_pass() -> MockLLM: + """返回始终 PASS 的 mock LLM。""" + return MockLLM([_make_llm_response("pass", "evidence found")]) + + +@pytest.fixture() +def mock_llm_fail() -> MockLLM: + """返回始终 FAIL 的 mock LLM。""" + return MockLLM([_make_llm_response("fail", "no evidence")]) + + +def _make_candidate() -> CandidateQuestion: + """构造测试用候选题目。""" + return CandidateQuestion( + question_id="q-001", + video_id="v-001", + task_type="Action Reasoning", + skill_target="M1", + question="What does the person do after picking up the book?", + options=( + "A. Reads it", + "B. Puts it back", + "C. Throws it away", + "D. Gives it to someone", + ), + answer="A", + source_nodes=("L2_001", "L3_001"), + difficulty="medium", + ) + + +def _make_tree() -> TreeIndex: + """构造最小测试树。""" + l3_card = L3Card( + frame_summary="Person picks up a book from the shelf and starts reading it.", + visible_entities=["person", "book", "shelf"], + ongoing_actions=["picking up book", "reading"], + visible_text=[], + spatial_layout="person in center, shelf on left", + visual_attributes={}, + subtitle="He picks up the book and reads.", + ) + l3 = L3Node(id="L3_001", card=l3_card) + + l2_card = L2Card( + event_description="A person picks up a book from the shelf and begins reading it attentively.", + entities=["person", "book"], + actions=["pick up", "read"], + action_subjects=["person"], + visible_text=[], + spatial_relations="person near shelf", + state_changes="book moves from shelf to hands", + subtitle="He picks up the book and reads.", + ) + l2 = L2Node(id="L2_001", card=l2_card, children=[l3]) + + l1_card = L1Card( + scene_summary="Library scene with a person browsing and reading books.", + main_setting="library", + key_entities=["person", "books", "shelf"], + main_actions=["browsing", "reading"], + topic_keywords=["library", "reading"], + visible_text=[], + temporal_flow="enter → browse → pick up → read", + ) + l1 = L1Node(id="L1_001", card=l1_card, children=[l2]) + + meta = IndexMeta(source_path="test.mp4", modality="video") + return TreeIndex(metadata=meta, roots=[l1]) + + +def _make_postprocess(verbatim_ratio: float = 0.1) -> PostprocessResult: + """构造测试用后处理结果。""" + return PostprocessResult( + options=( + "A. Reads it", + "B. Puts it back", + "C. Throws it away", + "D. Gives it to someone", + ), + answer="A", + referent_violations=[], + verbatim_ratio=verbatim_ratio, + has_time_anchor=False, + ) + + +# --------------------------------------------------------------------------- +# TestGateKeyVerify +# --------------------------------------------------------------------------- + + +class TestGateKeyVerify: + """key_verify 门:验证答案在来源素材中有证据支撑。""" + + @pytest.mark.asyncio() + async def test_pass_evidence(self, mock_llm_pass: MockLLM) -> None: + """LLM 确认有证据 → PASS。""" + candidate = _make_candidate() + tree = _make_tree() + result = await _gate_key_verify(candidate, tree, mock_llm_pass, session_id="test-session") + assert result.verdict == GateVerdict.PASS + assert mock_llm_pass.calls # 确认调用了 LLM + + @pytest.mark.asyncio() + async def test_fail_no_evidence(self, mock_llm_fail: MockLLM) -> None: + """LLM 判断无证据 → FAIL。""" + candidate = _make_candidate() + tree = _make_tree() + result = await _gate_key_verify(candidate, tree, mock_llm_fail, session_id="test-session") + assert result.verdict == GateVerdict.FAIL + assert "no evidence" in result.reason + + +# --------------------------------------------------------------------------- +# TestGateBlindAnswer +# --------------------------------------------------------------------------- + + +class TestGateBlindAnswer: + """blind_answer 门:无上下文答对 → FAIL(题目太简单/泄漏)。""" + + @pytest.mark.asyncio() + async def test_pass_wrong(self) -> None: + """LLM 在无上下文时答错 → PASS(题目确实需要视频信息)。""" + # LLM 返回 "pass" 表示它无法正确回答 + llm = MockLLM([_make_llm_response("pass", "cannot determine without context")]) + candidate = _make_candidate() + result = await _gate_blind_answer(candidate, llm, session_id="test-session") + assert result.verdict == GateVerdict.PASS + + @pytest.mark.asyncio() + async def test_fail_correct(self) -> None: + """LLM 在无上下文时答对 → FAIL(题目泄漏)。""" + llm = MockLLM([_make_llm_response("fail", "answer is obvious from options")]) + candidate = _make_candidate() + result = await _gate_blind_answer(candidate, llm, session_id="test-session") + assert result.verdict == GateVerdict.FAIL + + +# --------------------------------------------------------------------------- +# TestGateMultiTrue +# --------------------------------------------------------------------------- + + +class TestGateMultiTrue: + """multi_true 门:多选项正确 → FAIL。""" + + @pytest.mark.asyncio() + async def test_pass_single(self) -> None: + """只有一个正确选项 → PASS。""" + llm = MockLLM([_make_llm_response("pass", "only one correct answer")]) + candidate = _make_candidate() + tree = _make_tree() + result = await _gate_multi_true(candidate, tree, llm, session_id="test-session") + assert result.verdict == GateVerdict.PASS + + @pytest.mark.asyncio() + async def test_fail_multi(self) -> None: + """多个选项可被视为正确 → FAIL。""" + llm = MockLLM([_make_llm_response("fail", "options A and B are both plausible")]) + candidate = _make_candidate() + tree = _make_tree() + result = await _gate_multi_true(candidate, tree, llm, session_id="test-session") + assert result.verdict == GateVerdict.FAIL + + +# --------------------------------------------------------------------------- +# TestGateLeakTest +# --------------------------------------------------------------------------- + + +class TestGateLeakTest: + """leak_test 门:按家族模板执行泄漏探测。""" + + @pytest.mark.asyncio() + async def test_per_family_template(self) -> None: + """使用家族特定的 probe_template 调用 LLM。""" + llm = MockLLM([_make_llm_response("pass", "no shortcut detected")]) + candidate = _make_candidate() + result = await _gate_leak_test(candidate, RETRIEVAL_FAMILY, llm, session_id="test-session") + assert result.verdict == GateVerdict.PASS + # 验证 session_id 被正确传递 + assert llm.calls[0]["session_id"] == "test-session" + + +# --------------------------------------------------------------------------- +# TestRunGates +# --------------------------------------------------------------------------- + + +class TestRunGates: + """run_gates 编排:并发四门 + verbatim 短路。""" + + @pytest.mark.asyncio() + async def test_all_pass(self) -> None: + """四门全 PASS → GateReport.passed=True。""" + llm = MockLLM([_make_llm_response("pass", f"gate {i} ok") for i in range(4)]) + candidate = _make_candidate() + tree = _make_tree() + postprocess = _make_postprocess(verbatim_ratio=0.1) + + report = await run_gates( + candidate=candidate, + tree=tree, + llm=llm, + family_spec=RETRIEVAL_FAMILY, + postprocess=postprocess, + session_id="test-session", + ) + assert report.passed is True + assert report.reject_reason is None + + @pytest.mark.asyncio() + async def test_high_verbatim_shortcircuits(self) -> None: + """verbatim_ratio > 0.5 → key_verify 直接 FAIL,不调用 LLM。""" + llm = MockLLM([_make_llm_response("pass", "should not be called")]) + candidate = _make_candidate() + tree = _make_tree() + postprocess = _make_postprocess(verbatim_ratio=0.8) + + report = await run_gates( + candidate=candidate, + tree=tree, + llm=llm, + family_spec=RETRIEVAL_FAMILY, + postprocess=postprocess, + session_id="test-session", + ) + assert report.passed is False + assert report.key_verify.verdict == GateVerdict.FAIL + assert "verbatim" in report.key_verify.reason.lower()