refactor: remove block-sequential gate path and gate_block knob (algo #6)

config/train_videomme.yaml 同时收录待入库的实验配置变更(run_id v2 /
concurrency 32 / batch_size 40)。tests/integration/test_v3_contract_e2e.py
的 run_id 断言按 Task 5 显式契约同步修正(原断言依赖旧隐式实例注入)。
This commit is contained in:
2026-07-17 04:40:14 -04:00
parent 0b839937df
commit 8958eee11b
18 changed files with 322 additions and 791 deletions
@@ -125,7 +125,6 @@ class _FakeConfig:
gate_delta_min: float = 0.02
gate_lambda_dir: float = -3.0
gate_e_rollback: float = 10.0
gate_block: int = 4
gate_n_max: int = 40
gate_p_low: float = 0.1
gate_p_high: float = 0.9
+1 -1
View File
@@ -338,7 +338,7 @@ class TestInferenceUnitAggregationEndToEnd:
def _assert_all_persisted(self, log: HarnessLog, questions: list[GeneratedQuestion]) -> None:
"""逐题溯源保留:含被剔除的孤儿题在内,每题仍逐题落 predictions。"""
rows = log.query("SELECT * FROM predictions WHERE run_id = ?", ("test-run",))
rows = log.query("SELECT * FROM predictions WHERE run_id = ?", ("run-v3-contract",))
persisted = {r["question_id"] for r in rows}
assert "orphan_o" in persisted, "孤儿题未逐题落库(逐题溯源被破坏)"
assert persisted == {q.question_id for q in questions}, "逐题落库题数与输入不符"
@@ -1,8 +1,9 @@
"""tests/unit/test_gate_block_unit.py — gate 块实际执行路径按 unit 跑。
"""tests/unit/test_gate_unit_scope.py — gate 真实执行路径按 unit 口径跑。
针对 app/harness/validate.py::validate_skill_local真实 gate 执行路径
断言混格阶梯下 gate 块按 unit 口径运行baseline_cache 键含 unit_id
n_used unit 累加pair_block 折叠 AR pair逐题 predictions 仍溯源
迁移自块序贯版 test_gate_block_unit.py载体 validate_skill_localTask 6 删除
针对 app/harness/validate.py::validate_skills_concurrent连续并发 gate 真实路径
断言混格阶梯下 gate unit 口径运行baseline_cache 键含 unit_idn_used
unit 累加pair_block 折叠 AR pair逐题 predictions 仍溯源
核心算法保真 #5(信息阶梯 e-process 口径从 question_id 迁至 unit_id)。
"""
@@ -15,7 +16,7 @@ import pytest
from app.harness.gate_ladder import BaselineCache, skill_hash
from app.harness.inference import PREDICTIONS_SCHEMA, InferenceResult
from app.harness.log import HarnessLog
from app.harness.validate import _ladder_units, validate_skill_local
from app.harness.validate import GateSpec, _ladder_units, validate_skills_concurrent
from core.evolution import GateParams
from core.types import GeneratedQuestion
@@ -136,6 +137,35 @@ def _make_mock_run_inference(
return mock_fn, call_log
def _mk_spec(ladder: list[GeneratedQuestion]) -> GateSpec:
"""由混格阶梯题序构造单题型 GateSpecunits 经 _ladder_units 聚合)。"""
return GateSpec(
task_type="temporal",
target_file="temporal.md",
candidate_content="improved skill",
base_skill_content="baseline skill content",
units=tuple(_ladder_units(ladder)),
gate_run_prefix="step1_gate_test",
)
async def _run_gate(workspace: Path, spec: GateSpec, mock_fn, log: HarnessLog, cache, params):
"""跑单 spec 的 validate_skills_concurrent 并返回该题型的 outcome。"""
outcomes = await validate_skills_concurrent(
workspace_dir=workspace,
base_skills_version="v1",
specs=[spec],
gate_params=params,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
concurrency=8,
)
return outcomes[spec.task_type]
class TestLadderUnits:
"""_ladder_units:阶梯题序聚合为单元并保持信息阶梯序。"""
@@ -177,7 +207,7 @@ class TestLadderUnits:
@pytest.mark.asyncio
async def test_gate_n_used_counts_units_not_questions(tmp_path: Path) -> None:
"""混格阶梯(1 pair + 2 single)→ n_used=3 单元,非 4 题。"""
"""混格阶梯(1 pair + 2 single)→ n_used=3 单元,非 4 题(迁移自块序贯版)"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
cache = BaselineCache(workspace / "baseline_cache.json")
@@ -186,7 +216,7 @@ async def test_gate_n_used_counts_units_not_questions(tmp_path: Path) -> None:
# 基线全错、候选全对 → 3 单元齐翻 W=3
baseline = {"p1_o": False, "p1_m": False, "s0": False, "s1": False}
candidate = {"p1_o": True, "p1_m": True, "s0": True, "s1": True}
mock_fn, call_log = _make_mock_run_inference(log, baseline, candidate)
mock_fn, _ = _make_mock_run_inference(log, baseline, candidate)
accept_params = GateParams(
e_confirm=15.0,
@@ -197,30 +227,14 @@ async def test_gate_n_used_counts_units_not_questions(tmp_path: Path) -> None:
e_rollback=10.0,
)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="improved skill",
base_skill_content="baseline skill content",
ladder_items=ladder,
gate_params=accept_params,
gate_block=10,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
)
outcome = await _run_gate(workspace, _mk_spec(ladder), mock_fn, log, cache, accept_params)
# n_used 按 unit 计(3),W 按 unit 计(3
assert outcome.n_used == 3
assert outcome.w == 3
assert outcome.l == 0
# 证据行按 unit 口径(3 行)
# 证据行按 unit 口径(3 行)ladder_rank 沿阶梯序连续
assert len(outcome.evidence_rows) == 3
assert [r["ladder_rank"] for r in outcome.evidence_rows] == [0, 1, 2]
# baseline_cache 键含 unit_idpair 用 pair_id、single 用 question_id
s_hash = skill_hash("baseline skill content")
assert cache.get("temporal", s_hash, "p1", "p1") is False
@@ -235,84 +249,62 @@ async def test_gate_n_used_counts_units_not_questions(tmp_path: Path) -> None:
@pytest.mark.asyncio
async def test_gate_pair_partial_flip_not_counted(tmp_path: Path) -> None:
"""AR pair 候选仅单向翻(T,F)→单元仍错,W 不被单题污染。"""
"""AR pair 候选仅单向翻(T,F)→单元仍错,W 不被单题污染(迁移自块序贯版)
前缀逐单元判定下 2 单元小阶梯会在首单元 futility 早停观测不到 pair 语义
2 single 拉长阶梯4 单元中 3 single 翻转 W=3pair 不计入
candidate_acc = 3/4
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
cache = BaselineCache(workspace / "baseline_cache.json")
ladder = [*_pair("p1"), _single("s0")]
ladder = [*_pair("p1"), _single("s0"), _single("s1"), _single("s2")]
baseline = {"p1_o": False, "p1_m": False, "s0": False}
# pair 只翻一半(p1_o 对、p1_m 错)→ 单元 AND 仍错;s0 翻对
candidate = {"p1_o": True, "p1_m": False, "s0": True}
baseline = {"p1_o": False, "p1_m": False, "s0": False, "s1": False, "s2": False}
# pair 只翻一半(p1_o 对、p1_m 错)→ 单元 AND 仍错;singles 全翻对
candidate = {"p1_o": True, "p1_m": False, "s0": True, "s1": True, "s2": True}
mock_fn, _ = _make_mock_run_inference(log, baseline, candidate)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="improved skill",
base_skill_content="baseline skill content",
ladder_items=ladder,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=10,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
outcome = await _run_gate(
workspace, _mk_spec(ladder), mock_fn, log, cache, _DEFAULT_GATE_PARAMS
)
# 只有 s0 单元翻转,pair 单元不计 W(保真 #5:不被 P/Q 单题污染)
assert outcome.w == 1
# 只有 single 单元翻转,pair 单元不计 W(保真 #5:不被 P/Q 单题污染)
assert outcome.w == 3
assert outcome.l == 0
assert outcome.n_used == 2
# candidate_acc 分母按 unit2 单元,1 对)→ 0.5
assert outcome.candidate_acc == 0.5
assert outcome.n_used == 4
# candidate_acc 分母按 unit4 单元,1 对)→ 3/4
assert outcome.candidate_acc == 0.75
finally:
log.close()
@pytest.mark.asyncio
async def test_gate_baseline_cache_hit_by_unit(tmp_path: Path) -> None:
"""基线缓存按 unit_id 预填充 → 基线侧全命中不发起推理。"""
"""基线缓存按 unit_id 预填充 → 基线侧全命中不发起推理(迁移自块序贯版)
阶梯补长到 4 单元避免首单元 futility 早停覆盖 pair single 两种 unit
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
cache = BaselineCache(workspace / "baseline_cache.json")
ladder = [*_pair("p1"), _single("s0")]
ladder = [*_pair("p1"), _single("s0"), _single("s1"), _single("s2")]
s_hash = skill_hash("baseline skill content")
# 按 unit_id 预填充(pair→pair_idsingle→question_id),全错
cache.put("temporal", s_hash, "p1", "p1", False)
cache.put("temporal", s_hash, "p1", "s0", False)
for unit_id in ("p1", "s0", "s1", "s2"):
cache.put("temporal", s_hash, "p1", unit_id, False)
baseline = {"p1_o": False, "p1_m": False, "s0": False}
candidate = {"p1_o": True, "p1_m": True, "s0": True}
baseline = {"p1_o": False, "p1_m": False, "s0": False, "s1": False, "s2": False}
candidate = {"p1_o": True, "p1_m": True, "s0": True, "s1": True, "s2": True}
mock_fn, call_log = _make_mock_run_inference(log, baseline, candidate)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="improved skill",
base_skill_content="baseline skill content",
ladder_items=ladder,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=10,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
outcome = await _run_gate(
workspace, _mk_spec(ladder), mock_fn, log, cache, _DEFAULT_GATE_PARAMS
)
base_calls = [c for c in call_log if c["run_id"].endswith("_base")]
assert base_calls == [], "unit 键全命中不应发起基线推理"
assert outcome.n_used == 2
assert outcome.n_used == 4
finally:
log.close()
-2
View File
@@ -171,7 +171,6 @@ class _FakeConfig:
gate_delta_min: float = 0.02
gate_lambda_dir: float = -3.0
gate_e_rollback: float = 10.0
gate_block: int = 4
gate_n_max: int = 40
gate_p_low: float = 0.1
gate_p_high: float = 0.9
@@ -306,7 +305,6 @@ class TestFingerprintStructuralVsDecision:
"gate_delta_min",
"gate_lambda_dir",
"gate_e_rollback",
"gate_block",
"gate_n_max",
"gate_p_low",
"gate_p_high",
+8 -9
View File
@@ -50,7 +50,6 @@ def _valid_kwargs() -> dict:
"gate_delta_min": 0.02,
"gate_lambda_dir": -0.642,
"gate_e_rollback": 10.0,
"gate_block": 8,
"gate_n_max": 40,
"gate_p_low": 0.05,
"gate_p_high": 0.95,
@@ -378,16 +377,16 @@ class TestGateValidation:
with pytest.raises(ValueError, match="gate_lambda_dir"):
_validate(cfg)
def test_block_exceeds_n_max_rejected(self) -> None:
"""gate_block > gate_n_max 应抛出 ValueError"""
cfg = _make_config(gate_block=50, gate_n_max=40)
with pytest.raises(ValueError, match="gate_block"):
def test_n_max_zero_rejected(self) -> None:
"""gate_n_max <= 0 应抛出 ValueError(迁移自块序贯版 gate_block 校验)"""
cfg = _make_config(gate_n_max=0)
with pytest.raises(ValueError, match="gate_n_max"):
_validate(cfg)
def test_block_zero_rejected(self) -> None:
"""gate_block <= 0 应抛出 ValueError"""
cfg = _make_config(gate_block=0)
with pytest.raises(ValueError, match="gate_block"):
def test_n_max_negative_rejected(self) -> None:
"""gate_n_max 为负也应报错"""
cfg = _make_config(gate_n_max=-1)
with pytest.raises(ValueError, match="gate_n_max"):
_validate(cfg)
def test_p_low_exceeds_p_high_rejected(self) -> None:
-3
View File
@@ -327,7 +327,6 @@ class TestBuildOrLoadPoolsFrozen:
gate_delta_min=0.02,
gate_lambda_dir=-0.642,
gate_e_rollback=10.0,
gate_block=8,
gate_n_max=40,
gate_p_low=0.05,
gate_p_high=0.95,
@@ -881,7 +880,6 @@ class TestRunHoldoutEvalConfig:
gate_delta_min=0.02,
gate_lambda_dir=-0.642,
gate_e_rollback=10.0,
gate_block=8,
gate_n_max=40,
gate_p_low=0.05,
gate_p_high=0.95,
@@ -932,7 +930,6 @@ class TestRunHoldoutEvalConfig:
gate_delta_min=0.02,
gate_lambda_dir=-0.642,
gate_e_rollback=10.0,
gate_block=8,
gate_n_max=40,
gate_p_low=0.05,
gate_p_high=0.95,
-1
View File
@@ -840,7 +840,6 @@ class TestRunnerFactoryInjection:
"gate_delta_min": 0.02,
"gate_lambda_dir": -0.642,
"gate_e_rollback": 10.0,
"gate_block": 8,
"gate_n_max": 40,
"gate_p_low": 0.05,
"gate_p_high": 0.95,
+219 -256
View File
@@ -1,7 +1,9 @@
"""tests/unit/test_harness_validate.py — app/harness/validate.py 的单元测试。
覆盖:数据类型字段、materialize 物化与清理、async validate_skill_local
accept/reject/prefix 校验/INFRA 护栏/缓存命中/最后一块终态)。
覆盖:数据类型字段、materialize 物化与清理、async validate_skills_concurrent
accept/reject/prefix 校验/INFRA 护栏/缓存命中/题尽终态)。async 用例迁移自
块序贯版(validate_skill_localTask 6 删除):载体换连续并发 gate,语义断言
保留;前缀逐单元判定使早停点比旧块判定更早(见各用例 docstring 的数值推导)。
"""
from __future__ import annotations
@@ -14,10 +16,12 @@ from app.harness.gate_ladder import BaselineCache, skill_hash
from app.harness.inference import PREDICTIONS_SCHEMA, InferenceResult
from app.harness.log import HarnessLog
from app.harness.validate import (
GateSpec,
Probation,
ValidationOutcome,
_ladder_units,
materialize_candidate_skill,
validate_skill_local,
validate_skills_concurrent,
)
from core.evolution import GateParams, RejectedEdit
from core.types import GeneratedQuestion
@@ -150,7 +154,7 @@ def _make_mock_run_inference(
def _make_all_infra_mock(log: HarnessLog, stop_reason: str):
"""构建基线全 INFRA 的 mock:每 record 写指定 INFRA stop_reasonerror/parse_error)。
"""构建全 INFRA 的 mock:每 record 写指定 INFRA stop_reasonerror/parse_error)。
与真实推理一致——per-record DB stop_reason 与汇总 stop_reason_counts 同源;护栏
分子按 unit 从 DB 读(_infra_question_ids_from_db),故须真实落 DB。total 返回
@@ -199,6 +203,48 @@ def _make_all_infra_mock(log: HarnessLog, stop_reason: str):
return mock_fn, call_log
def _mk_spec(
questions: list[GeneratedQuestion],
*,
candidate_content: str = "candidate skill",
gate_run_prefix: str = "step1_gate_test",
) -> GateSpec:
"""由阶梯题序构造单题型 GateSpecunits 经 _ladder_units 聚合为阶梯序单元)。"""
return GateSpec(
task_type="temporal",
target_file="temporal.md",
candidate_content=candidate_content,
base_skill_content="baseline skill content",
units=tuple(_ladder_units(questions)),
gate_run_prefix=gate_run_prefix,
)
async def _run_single_spec(
workspace: Path,
spec: GateSpec,
mock_fn,
log: HarnessLog,
cache: BaselineCache,
params: GateParams,
gate_guard_err: float = 0.5,
) -> ValidationOutcome:
"""跑单 spec 的 validate_skills_concurrent 并返回该题型的 outcome。"""
outcomes = await validate_skills_concurrent(
workspace_dir=workspace,
base_skills_version="v1",
specs=[spec],
gate_params=params,
gate_guard_err=gate_guard_err,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
concurrency=8,
)
return outcomes[spec.task_type]
def test_infra_stop_reasons_single_source() -> None:
"""app 侧 INFRA_STOP_REASONS 复用 core 常量(同一对象),杜绝未来漂移(M-2)。"""
from app.harness import validate
@@ -331,13 +377,17 @@ class TestMaterializeCandidateSkill:
# ===========================================================================
# async 验证测试
# async 验证测试(迁移自块序贯版 validate_skill_local
# ===========================================================================
@pytest.mark.asyncio
async def test_validate_skill_local_accept(tmp_path: Path) -> None:
"""候选全对、基线全错 → 高 e 值 → accept_confirmed"""
async def test_validate_concurrent_accept(tmp_path: Path) -> None:
"""候选全对、基线全错 → 高 e 值 → accept_confirmed(迁移自块序贯版)。
6 单元连胜:E=(2^(W+1)-1)/(W+1),前 5 单元 E<15 且不触方向/futility
第 6 单元 E=18.14 ≥ e_confirm=15 → 与旧块判定同点收敛(W=6, n_used=6)。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
questions = _make_questions(6)
@@ -359,23 +409,13 @@ async def test_validate_skill_local_accept(tmp_path: Path) -> None:
)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="improved skill",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=accept_params,
gate_block=6,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
outcome = await _run_single_spec(
workspace,
_mk_spec(questions, candidate_content="improved skill"),
mock_fn,
log,
cache,
accept_params,
)
assert outcome.accepted is True
@@ -387,6 +427,8 @@ async def test_validate_skill_local_accept(tmp_path: Path) -> None:
assert outcome.candidate_acc == 1.0
assert outcome.baseline_acc == 0.0
assert len(outcome.evidence_rows) == 6
# 阶梯序前缀消费:ladder_rank 连续(替代旧块边界断言)
assert [r["ladder_rank"] for r in outcome.evidence_rows] == list(range(6))
# 终态证据行携带 stop_reason
assert outcome.evidence_rows[-1]["stop_reason"] == "confirmed"
# 候选临时目录应被清理
@@ -398,50 +440,45 @@ async def test_validate_skill_local_accept(tmp_path: Path) -> None:
@pytest.mark.asyncio
async def test_validate_skill_local_reject(tmp_path: Path) -> None:
"""候选全错、基线全对 → L 高 → 方向拒绝"""
async def test_validate_concurrent_reject_directional(tmp_path: Path) -> None:
"""候选全错、基线全对 → L 高 → 方向拒绝(迁移自块序贯版)。
前缀逐单元判定下早停点前移:15 单元阶梯保证 L=1..3 时 futility 不先触发
E(w+n_rem, l) ≥ 3),L=4 时 Wald=4·ln0.6=-2.04 ≤ lambda_dir=-2.0 →
directional 早停于第 4 单元(旧块版一次性判整块故 L=6)。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
questions = _make_questions(6)
questions = _make_questions(15)
cache = BaselineCache(workspace / "baseline_cache.json")
# 基线全对,候选全错 → W=0, L=6 → 方向拒绝
baseline_correct = {f"q{i}": True for i in range(6)}
candidate_correct = {f"q{i}": False for i in range(6)}
baseline_correct = {f"q{i}": True for i in range(15)}
candidate_correct = {f"q{i}": False for i in range(15)}
mock_fn, _ = _make_mock_run_inference(log, baseline_correct, candidate_correct)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="bad skill",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=6,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
outcome = await _run_single_spec(
workspace,
_mk_spec(questions, candidate_content="bad skill"),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
)
assert outcome.accepted is False
assert outcome.action == "reject"
assert outcome.stop_reason == "directional"
assert outcome.w == 0
assert outcome.l == 6
assert outcome.l == 4
assert outcome.n_used == 4
finally:
log.close()
@pytest.mark.asyncio
async def test_gate_prefix_must_contain_gate(tmp_path: Path) -> None:
"""gate_run_prefix 不含 '_gate_' 时抛 ValueError。"""
"""gate_run_prefix 不含 '_gate_' 时抛 ValueError(迁移自块序贯版)"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
questions = _make_questions(4)
@@ -452,23 +489,13 @@ async def test_gate_prefix_must_contain_gate(tmp_path: Path) -> None:
try:
with pytest.raises(ValueError, match="_gate_"):
await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="content",
base_skill_content="baseline",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=4,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=noop_fn,
log=log,
gate_run_prefix="step1_no_marker",
await _run_single_spec(
workspace,
_mk_spec(questions, gate_run_prefix="step1_no_marker"),
noop_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
)
finally:
log.close()
@@ -476,34 +503,26 @@ async def test_gate_prefix_must_contain_gate(tmp_path: Path) -> None:
@pytest.mark.asyncio
async def test_infra_guard_threshold(tmp_path: Path) -> None:
"""推理错误率超阈值时抛 RuntimeError护栏分子/分母 unit 同粒度)。"""
"""推理错误率超阈值时抛 RuntimeError迁移自块序贯版,分子/分母 unit 同粒度)。
12 个 single 双臂全 INFRA errorerrors 按单元去重逐单元 +1,分母逐臂 +1,
分母 ≥10 后错误率 >0.5 → 护栏熔断。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
# 需要 >=10 unit 分母才触发护栏:12 个 single,基线全 INFRA error。
# 首块全 INFRA → valid_chunk 空 → errors=12/denom=12=1.0>0.5 触发护栏。
questions = _make_questions(12)
cache = BaselineCache(workspace / "baseline_cache.json")
mock_fn, _ = _make_all_infra_mock(log, "error")
try:
with pytest.raises(RuntimeError, match="错误率过高"):
await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="content",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=12,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
await _run_single_spec(
workspace,
_mk_spec(questions),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
)
finally:
log.close()
@@ -511,7 +530,10 @@ async def test_infra_guard_threshold(tmp_path: Path) -> None:
@pytest.mark.asyncio
async def test_baseline_cache_hit(tmp_path: Path) -> None:
"""基线缓存全命中时不发起基线侧推理"""
"""基线缓存全命中时不发起基线侧推理(迁移自块序贯版)。
连续并发 gate 下候选侧逐单元发臂:4 单元 → 4 次 cand 调用(旧块版整块 1 次)。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
questions = _make_questions(4)
@@ -528,30 +550,20 @@ async def test_baseline_cache_hit(tmp_path: Path) -> None:
mock_fn, call_log = _make_mock_run_inference(log, baseline_correct, candidate_correct)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="improved skill",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=4,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
outcome = await _run_single_spec(
workspace,
_mk_spec(questions, candidate_content="improved skill"),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
)
# 只有候选侧调用了 run_inference(_cand),基线侧全命中不调用
base_calls = [c for c in call_log if c["run_id"].endswith("_base")]
cand_calls = [c for c in call_log if c["run_id"].endswith("_cand")]
assert len(base_calls) == 0, "基线缓存全命中不应发起推理"
assert len(cand_calls) == 1
assert len(cand_calls) == 4
assert outcome.accepted is True
finally:
log.close()
@@ -559,21 +571,21 @@ async def test_baseline_cache_hit(tmp_path: Path) -> None:
@pytest.mark.asyncio
async def test_baseline_infra_error_not_cached(tmp_path: Path) -> None:
"""基线臂 INFRA error 的 unit 不写入 BaselineCache(不永久污染),且从有效单元排除。"""
from app.harness.gate_ladder import skill_hash
from app.harness.question_units import build_units
from app.harness.validate import _resolve_baseline_block
"""基线臂 INFRA error 的 unit 不写入 BaselineCache(不永久污染),且从配对剔除。
迁移自块序贯版 _resolve_baseline_block 直测:改经 validate_skills_concurrent
端到端验证同一契约——INFRA 单元不落缓存、不入配对;干净单元正常缓存并消费。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
questions = _make_questions(2) # q0 干净, q1 INFRA error
units = build_units(questions)
questions = _make_questions(2) # q0 基线 INFRA error, q1 干净
cache = BaselineCache(workspace / "baseline_cache.json")
s_hash = skill_hash("baseline skill content")
async def mock_fn(qs, *, run_id, skills_dir):
is_base = run_id.endswith("_base")
for q in qs:
is_err = q.question_id == "q1"
is_err = is_base and q.question_id == "q0"
log.insert(
"predictions",
{
@@ -594,54 +606,49 @@ async def test_baseline_infra_error_not_cached(tmp_path: Path) -> None:
)
return InferenceResult(
run_id=run_id,
accuracy=0.5,
total=2,
correct=1,
accuracy=0.0,
total=len(qs),
correct=0,
per_task_type={},
steps_mean=1.0,
token_usage={"prompt_tokens": 10, "completion_tokens": 10},
stop_reason_counts={"completed": 1, "error": 1},
stop_reason_counts={},
)
try:
b_units, valid_units, _errors_inc, _denom_inc = await _resolve_baseline_block(
units=units,
task_type="temporal",
s_hash=s_hash,
prompts_version="p1",
baseline_cache=cache,
base_skills_dir=workspace / "skills" / "v1",
run_inference=mock_fn,
log=log,
run_id="step1_gate_b0_base",
outcome = await _run_single_spec(
workspace,
_mk_spec(questions),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
gate_guard_err=0.9, # 分母 <10 不触发错误率护栏
)
# q1 是 INFRA:不写缓存、不入 b_units、不在有效单元里
assert cache.get("temporal", s_hash, "p1", "q1") is None
assert "q1" not in b_units
assert all(u.unit_id != "q1" for u in valid_units)
# q0 干净:正常缓存并入 b_units/valid_units
assert cache.get("temporal", s_hash, "p1", "q0") is True
assert b_units["q0"] is True
assert any(u.unit_id == "q0" for u in valid_units)
# q0 是 INFRA:不写缓存、不入配对观测
assert cache.get("temporal", s_hash, "p1", "q0") is None
assert "q0" not in outcome.improvements + outcome.regressions
# q1 干净:正常缓存并被消费(唯一有效单元)
assert cache.get("temporal", s_hash, "p1", "q1") is True
assert outcome.n_used == 1
finally:
log.close()
@pytest.mark.asyncio
async def test_infra_guard_counts_units_not_records(tmp_path: Path) -> None:
"""护栏分子按 unit AR pair 两 record 全 INFRA 只计 1 个 INFRA unit(而非 2
async def test_infra_errors_counted_per_unit_not_per_record(tmp_path: Path) -> None:
"""护栏分子按 unit 去重AR pair 两 record、双臂全 INFRA 只计 1 个 error
回归 I-3:分子此前用 stop_reason_counts 逐 record 计数,分母 denom_inc=r.total
是 unit 粒度;AR pair(一 unit 两 record)致分子被放大、误触发 gate_guard_err。
分子改为"含 INFRA record 的 unit 数"后与分母同粒度(核心算法保真 #5/#6)。
迁移自块序贯版 _resolve_baseline_block 直测(回归 I-3):分子若逐 record /
逐臂计数会被放大(一 unit 两 record × 两臂 = 4),与 unit 粒度分母失配致
gate_guard_err 误触发。新载体 _run_unit_arm + _register_arm_arrival 按
slot.excluded() 去重(核心算法保真 #5/#6)。
"""
from app.harness.gate_ladder import skill_hash
from app.harness.question_units import build_units
from app.harness.validate import _resolve_baseline_block
from app.harness.validate import _GateRun, _QuestionSlots, _run_unit_arm
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
# 一个 AR pair(两成员共享 pair_id)→ build_units 折叠为 1 个 pair unit
common = {
"video_id": "vp",
"task_type": "temporal",
@@ -660,7 +667,17 @@ async def test_infra_guard_counts_units_not_records(tmp_path: Path) -> None:
units = build_units(pair)
assert len(units) == 1 # 前置:pair 折叠为 1 个 unit
cache = BaselineCache(workspace / "baseline_cache.json")
s_hash = skill_hash("baseline skill content")
run = _GateRun.from_spec(
GateSpec(
task_type="temporal",
target_file="temporal.md",
candidate_content="cand",
base_skill_content="baseline skill content",
units=tuple(units),
gate_run_prefix="step1_gate_test",
)
)
s_hash = run.s_hash
async def mock_fn(qs, *, run_id, skills_dir):
# 两 record 皆 INFRA error
@@ -683,7 +700,7 @@ async def test_infra_guard_counts_units_not_records(tmp_path: Path) -> None:
"steps_json": "[]",
},
)
# total 为 unit 粒度(1 个 pair unit);stop_reason_counts 为 record 粒度2
# total 为 unit 粒度(1 个 pair unit);record 粒度为 2
return InferenceResult(
run_id=run_id,
accuracy=0.0,
@@ -695,94 +712,57 @@ async def test_infra_guard_counts_units_not_records(tmp_path: Path) -> None:
stop_reason_counts={"error": 2},
)
slots = _QuestionSlots(4)
try:
_b_units, valid_units, errors_inc, denom_inc = await _resolve_baseline_block(
units=units,
task_type="temporal",
s_hash=s_hash,
prompts_version="p1",
baseline_cache=cache,
base_skills_dir=workspace / "skills" / "v1",
run_inference=mock_fn,
log=log,
run_id="step1_gate_b0_base",
)
# 分子按 unit 计:1 个 INFRA unit(不是 2 条 record);分母同粒度 = r.total = 1
assert errors_inc == 1
assert denom_inc == 1
# 整对 INFRA → 从有效单元剔除
assert valid_units == []
for arm in ("base", "cand"):
await _run_unit_arm(
run,
0,
arm,
slots,
mock_fn,
log,
cache,
"p1",
workspace / "skills" / "v1",
workspace / "skills" / "v1",
_DEFAULT_GATE_PARAMS,
0.9,
)
# 分子按 unit 去重:双臂 × 两 record 只计 1 个 error;分母按臂 total 累计 = 2
assert run.errors == 1
assert run.infra_denom == 2
assert run.slots[0].base_infra and run.slots[0].cand_infra
# INFRA 单元不写缓存
assert cache.get("temporal", s_hash, "p1", "p1") is None
finally:
log.close()
@pytest.mark.asyncio
async def test_all_infra_ladder_raises_clear_error(tmp_path: Path) -> None:
"""整个阶梯所有 unit 都被判为 INFRA 排除 → 明确 RuntimeError非误导性空阶梯断言)。"""
"""整个阶梯所有 unit 都被判为 INFRA 排除 → 明确 RuntimeError迁移自块序贯版)。
连续并发 gate 下双臂独立发射,候选臂不再依赖基线侧结果(旧版"全 INFRA 块
不空跑候选"的断言随块编排一并删除)。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
questions = _make_questions(4)
cache = BaselineCache(workspace / "baseline_cache.json")
candidate_calls: list[str] = []
async def mock_fn(qs, *, run_id, skills_dir):
if run_id.endswith("_cand"):
candidate_calls.append(run_id)
# 基线臂逐题全部 INFRA error(候选臂在修复后不应被空跑)
for q in qs:
log.insert(
"predictions",
{
"run_id": run_id,
"video_id": "v0",
"question_id": q.question_id,
"task_type": "temporal",
"prediction": "",
"answer": "A",
"evidence": "",
"reasoning": "",
"steps_used": 1,
"prompt_tokens": 10,
"completion_tokens": 10,
"stop_reason": "error",
"steps_json": "[]",
},
)
total = len(qs)
return InferenceResult(
run_id=run_id,
accuracy=0.0,
total=total,
correct=0,
per_task_type={},
steps_mean=1.0,
token_usage={"prompt_tokens": 10, "completion_tokens": 10},
stop_reason_counts={"error": total},
)
mock_fn, _ = _make_all_infra_mock(log, "error")
try:
with pytest.raises(RuntimeError, match="INFRA"):
await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="content",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=4,
gate_n_max=20,
gate_guard_err=0.9, # 高阈值:4 题 <10 分母不触发错误率护栏
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
await _run_single_spec(
workspace,
_mk_spec(questions),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
gate_guard_err=0.9, # 4 单元分母 <10 不触发错误率护栏 → 逼出全排除分支
)
# 全 INFRA 块不应触发候选空跑
assert candidate_calls == []
finally:
log.close()
@@ -792,42 +772,33 @@ async def test_parse_error_counts_toward_guard(tmp_path: Path) -> None:
"""stop_reason=parse_error 也计入护栏错误率(与 INFRA 判定口径一致)→ 超阈值熔断。"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
# 12 个 single,基线全 parse_errorper-record 落 DB,护栏按 unit 从 DB 读)。
# 首块全 INFRA → errors=12/denom=12=1.0>0.5 → parse_error 亦触发护栏。
questions = _make_questions(12)
cache = BaselineCache(workspace / "baseline_cache.json")
mock_fn, _ = _make_all_infra_mock(log, "parse_error")
try:
with pytest.raises(RuntimeError, match="错误率过高"):
await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="content",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=12,
gate_n_max=20,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
await _run_single_spec(
workspace,
_mk_spec(questions),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
)
finally:
log.close()
@pytest.mark.asyncio
async def test_last_block_terminal(tmp_path: Path) -> None:
"""单块 + n_remaining=0 → 终态判定(provisional 或 inertia),非 continue。"""
async def test_ladder_exhaustion_terminal(tmp_path: Path) -> None:
"""题尽(n_remaining=0→ 终态判定(provisional 或 inertia),非 continue。
迁移自块序贯版"最后一块终态":块边界不存在了,等价语义是阶梯耗尽时
第四出口兜底,终态行携带 stop_reason。
"""
workspace = _setup_workspace(tmp_path)
log = _make_log(workspace)
# 4 题,gate_block=4 → 一块走完,n_remaining=0
questions = _make_questions(4)
cache = BaselineCache(workspace / "baseline_cache.json")
@@ -837,23 +808,13 @@ async def test_last_block_terminal(tmp_path: Path) -> None:
mock_fn, _ = _make_mock_run_inference(log, baseline_correct, candidate_correct)
try:
outcome = await validate_skill_local(
workspace_dir=workspace,
base_skills_version="v1",
task_type="temporal",
target_file="temporal.md",
candidate_content="candidate skill",
base_skill_content="baseline skill content",
ladder_items=questions,
gate_params=_DEFAULT_GATE_PARAMS,
gate_block=4,
gate_n_max=4,
gate_guard_err=0.5,
baseline_cache=cache,
prompts_version="p1",
run_inference=mock_fn,
log=log,
gate_run_prefix="step1_gate_test",
outcome = await _run_single_spec(
workspace,
_mk_spec(questions),
mock_fn,
log,
cache,
_DEFAULT_GATE_PARAMS,
)
# n_remaining=0 → 不可能是 continue
@@ -865,6 +826,8 @@ async def test_last_block_terminal(tmp_path: Path) -> None:
"futility",
)
assert outcome.n_used == 4
# 阶梯序前缀消费:ladder_rank 连续
assert [r["ladder_rank"] for r in outcome.evidence_rows] == list(range(4))
# 终态行标记 stop_reason
assert outcome.evidence_rows[-1]["stop_reason"] != ""
finally:
@@ -64,7 +64,6 @@ def _base_config(workspace_dir: Path, store_dir: Path) -> RunConfig:
gate_delta_min=0.02,
gate_lambda_dir=-0.642,
gate_e_rollback=10.0,
gate_block=8,
gate_n_max=40,
gate_p_low=0.05,
gate_p_high=0.95,