"""连续并发 gate 编排测试:乱序到达/多题型隔离/终态组装/全 INFRA。""" from __future__ import annotations import asyncio from typing import TYPE_CHECKING import pytest if TYPE_CHECKING: from pathlib import Path from app.harness.gate_ladder import BaselineCache from app.harness.validate import GateSpec, validate_skills_concurrent from tests.unit.test_gate_prefix import _PARAMS, _mk_unit from tests.unit.test_gate_unit_arm import _FakeLog def _mk_spec(task_type: str, slug: str, n: int) -> GateSpec: """构造 n 个 single 单元的 gate 规格(unit_id 形如 -q)。""" return GateSpec( task_type=task_type, target_file=f"{slug}.md", candidate_content=f"cand-{slug}", base_skill_content=f"base-{slug}", units=tuple(_mk_unit(f"{slug}-q{i}", task_type) for i in range(n)), gate_run_prefix=f"r_e1_s0_gate_{slug}", ) def _scripted_inference(log: _FakeLog, script: dict[str, tuple[bool, float]]): """脚本化假推理:按 question_id+臂 决定 (对错, 延迟秒),制造乱序到达。""" class _R: def __init__(self, run_id: str, total: int) -> None: self.run_id = run_id self.total = total async def _run(questions, *, run_id: str, skills_dir: Path): arm = "cand" if run_id.endswith("_cand") else "base" correct, delay = script[f"{questions[0].question_id}|{arm}"] await asyncio.sleep(delay) for q in questions: log.rows.append( { "run_id": run_id, "question_id": q.question_id, "prediction": "A" if correct else "B", "answer": "A", "stop_reason": "finished", "steps_json": "[]", } ) return _R(run_id, len(questions)) return _run @pytest.mark.asyncio async def test_out_of_order_arrival_still_ladder_order(tmp_path, monkeypatch) -> None: """尾部先到、头部后到:判定结果与顺序到达完全相同(前缀有序性端到端)。""" spec = _mk_spec("Action Reasoning", "action-reasoning", 4) log = _FakeLog() script = {} for i in range(4): # 头部 q0 最慢;全部翻转为 W(base 错 cand 对) script[f"action-reasoning-q{i}|base"] = (False, 0.05 if i == 0 else 0.0) script[f"action-reasoning-q{i}|cand"] = (True, 0.05 if i == 0 else 0.0) monkeypatch.setattr( "app.harness.validate.materialize_candidate_skill", lambda *a, **k: tmp_path / "cand", ) outcomes = await validate_skills_concurrent( workspace_dir=tmp_path, base_skills_version="v1", specs=[spec], gate_params=_PARAMS, gate_guard_err=0.10, baseline_cache=BaselineCache(tmp_path / "bc.json"), prompts_version="v1", run_inference=_scripted_inference(log, script), log=log, concurrency=8, ) o = outcomes["Action Reasoning"] assert o.w == 4 and o.l == 0 assert [r["ladder_rank"] for r in o.evidence_rows] == [0, 1, 2, 3] @pytest.mark.asyncio async def test_two_types_isolated(tmp_path, monkeypatch) -> None: """两题型并行:计数互不污染,各自独立判定。 A 型 4 单元全 W(题尽 accept_provisional);B 型 2 单元全平 (futility 早停,W=L=0)——两型结果都不受对方污染。 """ spec_a = _mk_spec("Action Reasoning", "action-reasoning", 4) spec_b = _mk_spec("Counting Problem", "counting-problem", 2) log = _FakeLog() script = {} for i in range(4): script[f"action-reasoning-q{i}|base"] = (False, 0.0) script[f"action-reasoning-q{i}|cand"] = (True, 0.0) for i in range(2): script[f"counting-problem-q{i}|base"] = (True, 0.0) script[f"counting-problem-q{i}|cand"] = (True, 0.0) monkeypatch.setattr( "app.harness.validate.materialize_candidate_skill", lambda *a, **k: tmp_path / "cand", ) outcomes = await validate_skills_concurrent( workspace_dir=tmp_path, base_skills_version="v1", specs=[spec_a, spec_b], gate_params=_PARAMS, gate_guard_err=0.10, baseline_cache=BaselineCache(tmp_path / "bc.json"), prompts_version="v1", run_inference=_scripted_inference(log, script), log=log, concurrency=8, ) assert outcomes["Action Reasoning"].w == 4 assert outcomes["Counting Problem"].w == 0 assert outcomes["Counting Problem"].l == 0 @pytest.mark.asyncio async def test_all_infra_raises(tmp_path, monkeypatch) -> None: """全单元 INFRA:保留现行 RuntimeError 语义(检查推理基础设施)。""" spec = _mk_spec("Action Reasoning", "action-reasoning", 2) log = _FakeLog() class _R: def __init__(self, run_id, total): self.run_id, self.total = run_id, total async def _infra_run(questions, *, run_id, skills_dir): for q in questions: log.rows.append( { "run_id": run_id, "question_id": q.question_id, "prediction": "", "answer": "A", "stop_reason": "error", "steps_json": "[]", } ) return _R(run_id, len(questions)) monkeypatch.setattr( "app.harness.validate.materialize_candidate_skill", lambda *a, **k: tmp_path / "cand", ) with pytest.raises(RuntimeError): await validate_skills_concurrent( workspace_dir=tmp_path, base_skills_version="v1", specs=[spec], gate_params=_PARAMS, gate_guard_err=0.99, # 护栏放宽,逼出全 INFRA 分支 baseline_cache=BaselineCache(tmp_path / "bc.json"), prompts_version="v1", run_inference=_infra_run, log=log, concurrency=8, ) @pytest.mark.asyncio async def test_partial_materialize_failure_cleans_up(tmp_path, monkeypatch) -> None: """第 2 个题型物化失败:OSError 传播,且第 1 个已物化目录被清理不泄漏。""" spec_a = _mk_spec("Action Reasoning", "action-reasoning", 1) spec_b = _mk_spec("Counting Problem", "counting-problem", 1) made: list[Path] = [] def _mat(workspace_dir, base_skills_version, target_file, content): if made: # 第 2 次调用:模拟磁盘错误 raise OSError("第 2 个题型物化失败(模拟)") d = tmp_path / "cand_a" d.mkdir() made.append(d) return d monkeypatch.setattr("app.harness.validate.materialize_candidate_skill", _mat) async def _never_called(questions, *, run_id, skills_dir): raise AssertionError("物化失败后不应发起任何推理") with pytest.raises(OSError): await validate_skills_concurrent( workspace_dir=tmp_path, base_skills_version="v1", specs=[spec_a, spec_b], gate_params=_PARAMS, gate_guard_err=0.10, baseline_cache=BaselineCache(tmp_path / "bc.json"), prompts_version="v1", run_inference=_never_called, log=_FakeLog(), concurrency=8, ) assert len(made) == 1 assert not made[0].exists() @pytest.mark.asyncio async def test_guard_raise_cancels_remaining_tasks(tmp_path, monkeypatch) -> None: """护栏 raise 后其余在飞任务被取消收束:整体在超时内返回,不悬挂。 A 型 12 单元推理全 INFRA(stop_reason="error"),分母 ≥10 后错误率 1.0 超护栏 0.01 → RuntimeError;B 型推理挂在永不 set 的 Event 上,若无 取消收束,validate 将悬挂,wait_for 超时即为回归。 """ spec_a = _mk_spec("Action Reasoning", "action-reasoning", 12) spec_b = _mk_spec("Counting Problem", "counting-problem", 2) log = _FakeLog() hang = asyncio.Event() # 永不 set:B 型推理只能靠取消收束 class _R: def __init__(self, run_id, total): self.run_id, self.total = run_id, total async def _run(questions, *, run_id, skills_dir): if "counting-problem" in run_id: await hang.wait() for q in questions: log.rows.append( { "run_id": run_id, "question_id": q.question_id, "prediction": "", "answer": "A", "stop_reason": "error", "steps_json": "[]", } ) return _R(run_id, len(questions)) monkeypatch.setattr( "app.harness.validate.materialize_candidate_skill", lambda *a, **k: tmp_path / "cand", ) with pytest.raises(RuntimeError): await asyncio.wait_for( validate_skills_concurrent( workspace_dir=tmp_path, base_skills_version="v1", specs=[spec_a, spec_b], gate_params=_PARAMS, gate_guard_err=0.01, baseline_cache=BaselineCache(tmp_path / "bc.json"), prompts_version="v1", run_inference=_run, log=log, concurrency=8, ), timeout=5, )