"""记分板的测试:全部用合成日志,不跑模型、不起容器。 **重点是「构造出违反某条不变量的产物、记分板确实报击穿」那一组。** 一个永远返回通过的 判定器比没有判定器更糟——它会让一次什么都没验成的跑看起来全绿,而那正是最需要被看见的 情况。所以每条不变量都配一个反例,合法产物那组只是对照。 """ import json import re import sys from collections.abc import Callable, Sequence from pathlib import Path import pytest _REPO_ROOT = Path(__file__).resolve().parents[3] if str(_REPO_ROOT) not in sys.path: sys.path.insert(0, str(_REPO_ROOT)) from polyloop.serialization import encode # noqa: E402 from polyloop.types import ( # noqa: E402 ActionOutcome, ActionStatus, Intent, IntentKind, ModelCallResult, ModelReply, ReplayPolicy, RunFinished, RunResult, RunStarted, StepCompleted, StepRecord, StopReason, ) from tools.soak.scoreboard import ( # noqa: E402 _STOP_REASON_RULES, EXIT_BREACHED, EXIT_UNDETERMINED, Verdict, evaluate, main, render_report, ) #: 塞进观察与最终回答里的哨兵。压测语料里有第三方的真实文档,报告里一个字都不许出现。 SENTINEL = "SENTINEL-ZZZ-第三方文档正文" _TAGS = { RunStarted: "run_started", Intent: "intent", ModelCallResult: "model_call_result", StepCompleted: "step_completed", RunFinished: "run_finished", } # --------------------------------------------------------------------------- # 造合法产物 # --------------------------------------------------------------------------- #: 一步可以长成的几种样子,逐条对着库里造那条步记录的那个函数。 #: `action` 走 `_action_step`,`parse_failure` 走 `_parse_failure_step`, #: `call_failure` 走 `_failed_call_step`,`final_answer` 走 `_final_answer_step`。 STEP_KINDS = ( "action", "not_executed", "env_error", "parse_failure", "call_failure", "final_answer", ) def build_records( run_id: str, *, steps: int = 2, step_kinds: Sequence[str] | None = None, stop_reason: StopReason = StopReason.TASK_COMPLETED, max_steps: int = 5, max_actions: int = 20, max_parse_failures: int = 2, max_prompt_chars: int = 100000, model_replay_policy: ReplayPolicy = ReplayPolicy.NEVER, final_answer: str | None = None, complete_on_last: bool = True, tool_name: str | None = "run_code", sink_failures: int = 0, observation: str = "普通观察", prompt_chars_at: Callable[[int], int] = lambda index: 100 + index * 10, ) -> tuple[list[object], RunResult]: """造一份自洽的记录序列。 默认每一步都是「模型意图 / 模型结果 / 动作意图 / 逐步结果」那四条,动作执行成功。 `step_kinds` 给出的话就按它逐步造,取值见 `STEP_KINDS`——那几种步在库里由不同的函数 产出,字段形状各不相同,停止原因的自洽判据分的正是这些形状。 """ kinds = list(step_kinds) if step_kinds is not None else ["action"] * steps for kind in kinds: assert kind in STEP_KINDS, kind records: list[object] = [ RunStarted( run_id=run_id, parameter_snapshot={ "request.max_steps": str(max_steps), "request.max_actions": str(max_actions), "request.max_consecutive_parse_failures": str(max_parse_failures), "request.max_prompt_chars": str(max_prompt_chars), "store.kind": "jsonl", }, ) ] step_records: list[StepRecord] = [] for index, kind in enumerate(kinds): completed = complete_on_last and index == len(kinds) - 1 text = f"{observation}#{index}" chars = prompt_chars_at(index) records.append( Intent( run_id=run_id, kind=IntentKind.MODEL_CALL, call_index=index, result_id=f"m{index}", replay_policy=model_replay_policy, ) ) reply = ModelReply(call_id=f"c{index}", content="决策文本", thinking="") records.append( ModelCallResult( run_id=run_id, result_id=f"m{index}", # 模型调用失败那一步:结果记录在,但它记的是失败。 reply=None if kind == "call_failure" else reply, failure="TimeoutError: 网关没回" if kind == "call_failure" else None, ) ) common = { "step_idx": index, "content_chars": 4, "thinking_chars": 0, "observation_truncated_chars": 0, "prompt_chars": chars, "step_wall_ms": 7, } if kind == "call_failure": step = StepRecord( **common, # type: ignore[arg-type] raw_output="", action=None, parse_ok=False, # 这一步压根没走到解释器,所以没有回喂给模型的说明——这正是库用来把它和 # 解析失败分开的那一对字段。 parse_error=None, observation="[模型调用失败]", observation_is_synthetic=True, call_id=None, ) elif kind == "parse_failure": step = StepRecord( **common, # type: ignore[arg-type] raw_output="决策文本", action=None, parse_ok=False, parse_error="解释不出有效决策,请重新输出一个 JSON 对象。", observation="解释不出有效决策,请重新输出一个 JSON 对象。", observation_is_synthetic=True, call_id=f"c{index}", ) elif kind == "final_answer": step = StepRecord( **common, # type: ignore[arg-type] raw_output="决策文本", action=None, parse_ok=True, parse_error=None, observation="", observation_is_synthetic=False, call_id=f"c{index}", ) else: status = { "action": ActionStatus.EXECUTED, "not_executed": ActionStatus.NOT_EXECUTED, "env_error": ActionStatus.ENV_ERROR, }[kind] passthrough = status is ActionStatus.EXECUTED records.append( Intent( run_id=run_id, kind=IntentKind.ACTION, call_index=index, result_id=f"a{index}", replay_policy=ReplayPolicy.NEVER, ) ) outcome = ActionOutcome( status=status, observation=text, observation_is_synthetic=status is ActionStatus.NOT_EXECUTED, env_reported_completion=completed, observation_truncated_chars=0, ) step = StepRecord( **common, # type: ignore[arg-type] raw_output="决策文本", action="run_code", parse_ok=True, parse_error=None, # 未执行与环境故障两档,库换掉回填进历史的那段观察并把合成标记立起来。 observation=text if passthrough else "[动作没有执行]", observation_is_synthetic=not passthrough, call_id=f"c{index}", tool_name=tool_name, tool_arguments="{}", action_status=status, env_reported_completion=completed, ) step_records.append(step) records.append( StepCompleted( run_id=run_id, result_id=f"a{index}", action_outcome=outcome, step=step ) ) continue step_records.append(step) records.append(StepCompleted(run_id=run_id, result_id=None, action_outcome=None, step=step)) result = RunResult( run_id=run_id, stop_reason=stop_reason, final_answer=final_answer, steps=tuple(step_records), event_delivery_failures=sink_failures, ) records.append(RunFinished(run_id=run_id, result=result)) return records, result def write_log(path: Path, records: Sequence[object]) -> None: lines = [ json.dumps({"record": _TAGS[type(record)], **encode(record)}, ensure_ascii=False) for record in records ] path.write_text("\n".join(lines) + "\n", encoding="utf-8") def materialize( runs_dir: Path, run_id: str = "soak-0001", *, write_result: bool = True, write_events: bool = True, write_meta: bool = True, events: int | None = None, meta_overrides: dict[str, object] | None = None, **kwargs: object, ) -> RunResult: """把一个自洽的 run 的四个文件都写到磁盘上,返回它的运行结果。""" runs_dir.mkdir(parents=True, exist_ok=True) records, result = build_records(run_id, **kwargs) # type: ignore[arg-type] write_log(runs_dir / f"{run_id}.jsonl", records) if write_result: (runs_dir / f"{run_id}.result.json").write_text( json.dumps(encode(result), ensure_ascii=False), encoding="utf-8" ) if write_events: count = len(result.steps) if events is None else events payload = "".join( json.dumps({"kind": "step_finished", "run_id": run_id, "step_idx": index}) + "\n" for index in range(count) ) (runs_dir / f"{run_id}.events.jsonl").write_text(payload, encoding="utf-8") if write_meta: meta: dict[str, object] = { "scenario": "appworld", "task_id": "82e2fac_1", "phase": None, "wall_ms": 1234, "model_calls": len(result.steps), "sink_failures": result.event_delivery_failures, "env_executions": len(result.steps), "fault": None, "success": True, "resumed_from_step": None, } meta.update(meta_overrides or {}) (runs_dir / f"{run_id}.meta.json").write_text( json.dumps(meta, ensure_ascii=False), encoding="utf-8" ) return result def edit_log(path: Path, edit: Callable[[list[dict]], list[dict]]) -> None: """按行读出来、交给 `edit` 改、再写回去。制造反例用的。""" payloads = [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] changed = edit(payloads) path.write_text( "".join(json.dumps(item, ensure_ascii=False) + "\n" for item in changed), encoding="utf-8", ) def reshape_step( path: Path, *, step_idx: int, status: str, outcome_observation: str, step_observation: str, outcome_is_synthetic: bool = False, step_is_synthetic: bool = False, outcome_truncated: int = 0, step_truncated: int = 0, ) -> None: """把某一步改写成「执行器那侧与步记录那侧的观察不是同一段」的样子。 `build_records` 造出来的每一步都是 executed 且两侧逐字相同,而真实产物里被拒绝的动作 与环境故障那两档不长这样——库会替换回填进历史的观察。反例都从这里造。 """ def edit(items: list[dict]) -> list[dict]: for item in items: if item["record"] != "step_completed" or item["step"]["step_idx"] != step_idx: continue item["action_outcome"]["status"] = status item["action_outcome"]["observation"] = outcome_observation item["action_outcome"]["observation_is_synthetic"] = outcome_is_synthetic item["action_outcome"]["observation_truncated_chars"] = outcome_truncated item["step"]["action_status"] = status item["step"]["observation"] = step_observation item["step"]["observation_is_synthetic"] = step_is_synthetic item["step"]["observation_truncated_chars"] = step_truncated return items edit_log(path, edit) def edit_result_steps(runs_dir: Path, run_id: str, mutate: Callable[[list[dict]], None]) -> None: """同时改 run_finished 内嵌的那份结果与 `.result.json` 里的步。 两边一起改,「跨进程的结果与内存里的一致」才不会跟着一起红——这里要看的是停止原因那条, 不是那条。日志里独立的 step_completed 行不动:没有任何判据拿它和结果里的步对比。 """ def edit(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "run_finished": mutate(item["result"]["steps"]) return items edit_log(runs_dir / f"{run_id}.jsonl", edit) path = runs_dir / f"{run_id}.result.json" payload = json.loads(path.read_text(encoding="utf-8")) mutate(payload["steps"]) path.write_text(json.dumps(payload, ensure_ascii=False), encoding="utf-8") def append_dangling_intent( runs_dir: Path, run_id: str, *, replay_policy: str = "never", kind: str = "model_call" ) -> None: """在日志末尾补一条没有归宿的意图,模拟「意图写了、结果没写」那个断点。""" edit_log( runs_dir / f"{run_id}.jsonl", lambda items: [ *items, { "record": "intent", "run_id": run_id, "kind": kind, "call_index": 99, "result_id": "dangling-99", "replay_policy": replay_policy, }, ], ) def verdict_of(scoreboard, name: str) -> Verdict: for item in scoreboard.invariants: if item.name == name: return item.verdict raise AssertionError(f"记分板里没有名为 {name!r} 的不变量") def assert_breached(scoreboard, name: str) -> None: assert verdict_of(scoreboard, name) is Verdict.BREACHED, ( f"{name} 应当被击穿,实际 {verdict_of(scoreboard, name)}" ) result = next(item for item in scoreboard.invariants if item.name == name) for evidence in result.breaches: assert evidence.expected is not None assert evidence.actual is not None # --------------------------------------------------------------------------- # 合法产物:全部通过 # --------------------------------------------------------------------------- def test_legal_run_passes_every_invariant(tmp_path: Path) -> None: materialize(tmp_path) scoreboard = evaluate(tmp_path) assert scoreboard.verdict is Verdict.PASSED, [ (item.name, item.verdict, [e.describe() for e in item.breaches + item.undetermined]) for item in scoreboard.invariants if item.verdict is not Verdict.PASSED ] assert scoreboard.notes == () def test_several_legal_runs_pass(tmp_path: Path) -> None: materialize(tmp_path, "soak-0001", steps=3) materialize( tmp_path, "soak-0002", steps=5, stop_reason=StopReason.STEP_BUDGET, max_steps=5, complete_on_last=False, ) materialize( tmp_path, "soak-0003", # 两步,不是一步:一步凑不出相邻的两个 prompt_chars,那条会报无法判定。 steps=2, stop_reason=StopReason.AGENT_FINISHED, complete_on_last=False, final_answer="给出的答案", ) scoreboard = evaluate(tmp_path) assert scoreboard.verdict is Verdict.PASSED assert len(scoreboard.summaries) == 3 assert scoreboard.overall_stats.stop_reasons == { "agent_finished": 1, "step_budget": 1, "task_completed": 1, } def test_completing_tool_is_supplied_by_the_caller(tmp_path: Path) -> None: """task_completed 而最后一步没有环境侧证据:给了工具名就判得了,不给就是无法判定。""" materialize(tmp_path, complete_on_last=False, tool_name="finish_task") assert verdict_of(evaluate(tmp_path), "停止原因与轨迹自洽") is Verdict.UNDETERMINED assert ( verdict_of( evaluate(tmp_path, completing_tools=frozenset({"finish_task"})), "停止原因与轨迹自洽", ) is Verdict.PASSED ) assert ( verdict_of( evaluate(tmp_path, completing_tools=frozenset({"别的工具"})), "停止原因与轨迹自洽", ) is Verdict.BREACHED ) # --------------------------------------------------------------------------- # 逐条不变量的反例 # --------------------------------------------------------------------------- def test_unreadable_line_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) path = tmp_path / "soak-0001.jsonl" lines = path.read_text(encoding="utf-8").splitlines() lines[2] = "{这不是 JSON" path.write_text("\n".join(lines) + "\n", encoding="utf-8") scoreboard = evaluate(tmp_path) assert_breached(scoreboard, "日志能被读回来") assert any("第 3 行" in e.describe() for e in scoreboard.breached[0].breaches) def test_unknown_record_tag_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [{**items[0], "record": "谁知道这是什么"}] + items[1:], ) assert_breached(evaluate(tmp_path), "日志能被读回来") def test_torn_tail_is_reported_but_not_a_breach(tmp_path: Path) -> None: materialize(tmp_path) path = tmp_path / "soak-0001.jsonl" # 最后一行只写了一半、没有换行终结:进程被杀在写入中途的样子。 path.write_text(path.read_text(encoding="utf-8") + '{"record": "step_comp', encoding="utf-8") scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "日志能被读回来") is Verdict.PASSED assert any("撕裂尾行" in note for note in scoreboard.notes) assert "撕裂尾行" in render_report(scoreboard) def test_torn_line_in_the_middle_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) path = tmp_path / "soak-0001.jsonl" lines = path.read_text(encoding="utf-8").splitlines() # 被换行终结、却是半截 JSON:追加写不会产生这种东西,所以是损坏。 lines[3] = lines[3][: len(lines[3]) // 2] path.write_text("\n".join(lines) + "\n", encoding="utf-8") scoreboard = evaluate(tmp_path) assert_breached(scoreboard, "日志能被读回来") assert not any("撕裂尾行" in note for note in scoreboard.notes) def test_result_file_disagreeing_with_the_log_is_a_breach(tmp_path: Path) -> None: result = materialize(tmp_path) payload = encode(result) payload["final_answer"] = "跨进程读回来的和内存里的不一样" (tmp_path / "soak-0001.result.json").write_text( json.dumps(payload, ensure_ascii=False), encoding="utf-8" ) assert_breached(evaluate(tmp_path), "跨进程的结果与内存里的一致") def test_result_file_step_field_disagreeing_is_a_breach(tmp_path: Path) -> None: result = materialize(tmp_path) payload = encode(result) payload["steps"][1]["step_wall_ms"] = 99999 # type: ignore[index] (tmp_path / "soak-0001.result.json").write_text( json.dumps(payload, ensure_ascii=False), encoding="utf-8" ) scoreboard = evaluate(tmp_path) assert_breached(scoreboard, "跨进程的结果与内存里的一致") evidence = next( item for item in scoreboard.invariants if item.name == "跨进程的结果与内存里的一致" ) assert any("steps[1].step_wall_ms" in e.describe() for e in evidence.breaches) def test_step_index_gap_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, steps=3) def bump(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed" and item["step"]["step_idx"] == 1: item["step"]["step_idx"] = 2 return items edit_log(tmp_path / "soak-0001.jsonl", bump) assert_breached(evaluate(tmp_path), "步号连续") def test_dangling_intent_in_the_middle_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, steps=3) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [ item for item in items if not (item["record"] == "model_call_result" and item["result_id"] == "m0") ], ) assert_breached(evaluate(tmp_path), "意图都有归宿") def test_two_dangling_intents_are_a_breach(tmp_path: Path) -> None: materialize(tmp_path, steps=3) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [ item for item in items if not (item["record"] == "model_call_result" and item["result_id"] in {"m1", "m2"}) ], ) assert_breached(evaluate(tmp_path), "意图都有归宿") def test_one_dangling_intent_at_the_tail_is_the_crash_point_and_passes(tmp_path: Path) -> None: materialize(tmp_path, steps=2) def crash(items: list[dict]) -> list[dict]: kept = [item for item in items if item["record"] != "run_finished"] kept.append( { "record": "intent", "run_id": "soak-0001", "kind": "model_call", "call_index": 2, "result_id": "m2", "replay_policy": "never", } ) return kept edit_log(tmp_path / "soak-0001.jsonl", crash) assert verdict_of(evaluate(tmp_path), "意图都有归宿") is Verdict.PASSED def test_step_completed_pairing_violation_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) def unpair(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed" and item["step"]["step_idx"] == 0: item["result_id"] = None return items edit_log(tmp_path / "soak-0001.jsonl", unpair) assert_breached(evaluate(tmp_path), "步记录的内部不变量") def test_outcome_disagreeing_with_step_record_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) def skew(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed" and item["step"]["step_idx"] == 0: item["step"]["action_status"] = "not_executed" return items edit_log(tmp_path / "soak-0001.jsonl", skew) assert_breached(evaluate(tmp_path), "动作结果与步记录一致") def test_executed_step_with_a_different_observation_is_a_breach(tmp_path: Path) -> None: """executed 档是原样透传,两侧观察必须逐字相同。""" materialize(tmp_path) reshape_step( tmp_path / "soak-0001.jsonl", step_idx=0, status="executed", outcome_observation="环境返回的原文", step_observation="换了一段别的", ) assert_breached(evaluate(tmp_path), "动作结果与步记录一致") @pytest.mark.parametrize( ("status", "executor_text", "executor_is_synthetic"), [ ( "not_executed", "工具不存在:'final_answer',本次可见的是 " "['read_document', 'grep_document', 'write_note']", True, ), ("env_error", "容器没了:connection refused", False), ], ) def test_replaced_observation_is_not_a_breach( tmp_path: Path, status: str, executor_text: str, executor_is_synthetic: bool ) -> None: """未执行与环境故障两档,库刻意换掉回填进历史的观察,两侧文本不同是正常的。 数据形状照 `tools/soak/runs/full/govdoc-15-execute.jsonl` 第 36 行那条真实记录造: 执行器那侧留的是「工具不存在」的原文,步记录那侧是合成的那段提示。截断数也一起换掉 (库在这两档下一律填 0),所以它同样不该被比对。 """ materialize(tmp_path) reshape_step( tmp_path / "soak-0001.jsonl", step_idx=0, status=status, outcome_observation=executor_text, step_observation=( "[动作被拒绝,这一步没有执行任何工具]\n" "请对照工具清单检查工具名与参数,然后重新输出一个 JSON 对象。" ), outcome_is_synthetic=executor_is_synthetic, step_is_synthetic=True, outcome_truncated=40, step_truncated=0, ) assert verdict_of(evaluate(tmp_path), "动作结果与步记录一致") is Verdict.PASSED @pytest.mark.parametrize("status", ["not_executed", "env_error"]) def test_replaced_observation_without_the_synthetic_flag_is_a_breach( tmp_path: Path, status: str ) -> None: """库既然替换了观察,就必须把 observation_is_synthetic 立起来,不立才是真出了问题。""" materialize(tmp_path) reshape_step( tmp_path / "soak-0001.jsonl", step_idx=0, status=status, outcome_observation="执行器给的原文", step_observation="库换上去的那一段", outcome_is_synthetic=False, step_is_synthetic=False, ) assert_breached(evaluate(tmp_path), "动作结果与步记录一致") @pytest.mark.parametrize("status", ["not_executed", "env_error"]) def test_status_still_has_to_agree_in_every_branch(tmp_path: Path, status: str) -> None: """状态与完成标记这两项三档下都比:观察分档,它们不分。""" materialize(tmp_path) reshape_step( tmp_path / "soak-0001.jsonl", step_idx=0, status=status, outcome_observation="执行器给的原文", step_observation="库换上去的那一段", outcome_is_synthetic=True, step_is_synthetic=True, ) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [ {**item, "step": {**item["step"], "action_status": "executed"}} if item["record"] == "step_completed" and item["step"]["step_idx"] == 0 else item for item in items ], ) assert_breached(evaluate(tmp_path), "动作结果与步记录一致") def test_prompt_chars_going_backwards_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, steps=3) def shrink(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed" and item["step"]["step_idx"] == 2: item["step"]["prompt_chars"] = 1 return items edit_log(tmp_path / "soak-0001.jsonl", shrink) assert_breached(evaluate(tmp_path), "提示词字符数单调不减") def test_event_count_mismatch_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, steps=3, events=2) assert_breached(evaluate(tmp_path), "事件条数等于本进程走完的步数") def test_resumed_run_subtracts_the_skipped_steps(tmp_path: Path) -> None: """续跑:日志里有四步,本进程只真的走完后两步,所以只该有两条事件。""" materialize( tmp_path, steps=4, events=2, meta_overrides={"resumed_from_step": 2, "model_calls": 2} ) assert verdict_of(evaluate(tmp_path), "事件条数等于本进程走完的步数") is Verdict.PASSED materialize( tmp_path, "soak-0002", steps=4, events=4, meta_overrides={"resumed_from_step": 2}, ) assert_breached(evaluate(tmp_path), "事件条数等于本进程走完的步数") def test_delivery_failure_count_mismatch_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, sink_failures=0, meta_overrides={"sink_failures": 3}) assert_breached(evaluate(tmp_path), "投递失败计数对得上") def test_crosstalk_in_the_log_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) def swap(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed" and item["step"]["step_idx"] == 1: item["run_id"] = "soak-9999" return items edit_log(tmp_path / "soak-0001.jsonl", swap) assert_breached(evaluate(tmp_path), "不串台") def test_crosstalk_in_the_events_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path) (tmp_path / "soak-0001.events.jsonl").write_text( json.dumps({"kind": "step_finished", "run_id": "soak-0001", "step_idx": 0}) + "\n" + json.dumps({"kind": "step_finished", "run_id": "soak-9999", "step_idx": 1}) + "\n", encoding="utf-8", ) assert_breached(evaluate(tmp_path), "不串台") def test_task_completed_without_evidence_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, complete_on_last=False, tool_name="run_code") scoreboard = evaluate(tmp_path, completing_tools=frozenset()) assert_breached(scoreboard, "停止原因与轨迹自洽") def test_step_budget_not_matching_max_steps_is_a_breach(tmp_path: Path) -> None: materialize( tmp_path, steps=3, max_steps=9, stop_reason=StopReason.STEP_BUDGET, complete_on_last=False, ) assert_breached(evaluate(tmp_path), "停止原因与轨迹自洽") def test_agent_finished_without_final_answer_is_a_breach(tmp_path: Path) -> None: materialize( tmp_path, stop_reason=StopReason.AGENT_FINISHED, complete_on_last=False, final_answer="", ) assert_breached(evaluate(tmp_path), "停止原因与轨迹自洽") def test_cancelled_without_run_finished_is_a_breach(tmp_path: Path) -> None: materialize(tmp_path, stop_reason=StopReason.CANCELLED, complete_on_last=False) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [item for item in items if item["record"] != "run_finished"], ) assert_breached(evaluate(tmp_path), "停止原因与轨迹自洽") RULE = "停止原因与轨迹自洽" def test_parse_failed_repeatedly_tail_matches_the_limit(tmp_path: Path) -> None: materialize( tmp_path, step_kinds=["action", "parse_failure", "parse_failure"], stop_reason=StopReason.PARSE_FAILED_REPEATEDLY, max_parse_failures=2, complete_on_last=False, ) assert verdict_of(evaluate(tmp_path), RULE) is Verdict.PASSED def test_parse_failed_repeatedly_with_a_short_tail_is_a_breach(tmp_path: Path) -> None: """末尾只有两步解析失败,上限却是三——那个计数撞线时不可能停在两步。""" materialize( tmp_path, step_kinds=["action", "parse_failure", "parse_failure"], stop_reason=StopReason.PARSE_FAILED_REPEATEDLY, max_parse_failures=3, complete_on_last=False, ) assert_breached(evaluate(tmp_path), RULE) def test_parse_failed_repeatedly_tail_touching_the_env_is_a_breach(tmp_path: Path) -> None: """解析失败那一支根本不碰环境,末尾那几步不该有动作状态。""" materialize( tmp_path, step_kinds=["action", "parse_failure"], stop_reason=StopReason.PARSE_FAILED_REPEATEDLY, max_parse_failures=1, complete_on_last=False, ) def touch_env(steps: list[dict]) -> None: steps[-1]["action_status"] = "executed" edit_result_steps(tmp_path, "soak-0001", touch_env) assert_breached(evaluate(tmp_path), RULE) def test_context_overflow_with_every_step_inside_the_limit_passes(tmp_path: Path) -> None: materialize( tmp_path, steps=2, stop_reason=StopReason.CONTEXT_OVERFLOW, max_prompt_chars=1000, complete_on_last=False, ) assert verdict_of(evaluate(tmp_path), RULE) is Verdict.PASSED def test_context_overflow_with_a_step_over_the_limit_is_a_breach(tmp_path: Path) -> None: """超限的那次装配根本不产生步记录,所以落盘的每一步必定在线内。""" materialize( tmp_path, steps=2, stop_reason=StopReason.CONTEXT_OVERFLOW, max_prompt_chars=105, complete_on_last=False, ) assert_breached(evaluate(tmp_path), RULE) def test_context_overflow_without_any_step_is_undetermined(tmp_path: Path) -> None: """首次装配就超限的运行一步都没落盘,那是这个原因最典型的形态,可确实没东西可验。""" materialize( tmp_path, steps=0, stop_reason=StopReason.CONTEXT_OVERFLOW, complete_on_last=False, ) assert verdict_of(evaluate(tmp_path), RULE) is Verdict.UNDETERMINED def test_action_budget_counts_only_executed_steps(tmp_path: Path) -> None: """未执行的那一步不加已执行动作计数,所以两步里只有一步算数。""" materialize( tmp_path, step_kinds=["action", "not_executed"], stop_reason=StopReason.ACTION_BUDGET, max_actions=1, complete_on_last=False, ) assert verdict_of(evaluate(tmp_path), RULE) is Verdict.PASSED def test_action_budget_not_matching_max_actions_is_a_breach(tmp_path: Path) -> None: materialize( tmp_path, steps=2, stop_reason=StopReason.ACTION_BUDGET, max_actions=5, complete_on_last=False, ) assert_breached(evaluate(tmp_path), RULE) def test_env_error_with_a_broken_last_step_passes(tmp_path: Path) -> None: materialize( tmp_path, step_kinds=["action", "env_error"], stop_reason=StopReason.ENV_ERROR, complete_on_last=False, ) assert verdict_of(evaluate(tmp_path), RULE) is Verdict.PASSED def test_env_error_without_a_broken_last_step_is_a_breach(tmp_path: Path) -> None: materialize( tmp_path, step_kinds=["env_error", "action"], stop_reason=StopReason.ENV_ERROR, complete_on_last=False, ) assert_breached(evaluate(tmp_path), RULE) def test_llm_error_with_a_failed_call_step_passes(tmp_path: Path) -> None: materialize( tmp_path, step_kinds=["action", "call_failure"], stop_reason=StopReason.LLM_ERROR, complete_on_last=False, ) assert verdict_of(evaluate(tmp_path), RULE) is Verdict.PASSED def test_llm_error_ending_on_a_parse_failure_is_a_breach(tmp_path: Path) -> None: """解析失败那一步也没有动作结果,两者只能靠 parse_error 分开——它必须为空。""" materialize( tmp_path, step_kinds=["action", "parse_failure"], stop_reason=StopReason.LLM_ERROR, complete_on_last=False, ) scoreboard = evaluate(tmp_path) assert_breached(scoreboard, RULE) rule = next(item for item in scoreboard.invariants if item.name == RULE) assert any("parse_error 有值" in e.describe() for e in rule.breaches) def test_llm_error_ending_on_an_action_is_a_breach(tmp_path: Path) -> None: materialize( tmp_path, steps=2, stop_reason=StopReason.LLM_ERROR, complete_on_last=False, ) assert_breached(evaluate(tmp_path), RULE) def test_resume_state_unknown_with_a_never_intent_dangling_passes(tmp_path: Path) -> None: materialize( tmp_path, steps=1, stop_reason=StopReason.RESUME_STATE_UNKNOWN, complete_on_last=False, ) append_dangling_intent(tmp_path, "soak-0001", replay_policy="never") assert verdict_of(evaluate(tmp_path), RULE) is Verdict.PASSED def test_resume_state_unknown_with_a_safe_intent_is_a_breach(tmp_path: Path) -> None: """声明可安全重放的意图会被直接重放,不会停在这一档。""" materialize( tmp_path, steps=1, stop_reason=StopReason.RESUME_STATE_UNKNOWN, complete_on_last=False, ) append_dangling_intent(tmp_path, "soak-0001", replay_policy="safe") assert_breached(evaluate(tmp_path), RULE) def test_resume_state_unknown_without_a_dangling_intent_is_a_breach(tmp_path: Path) -> None: materialize( tmp_path, steps=1, stop_reason=StopReason.RESUME_STATE_UNKNOWN, complete_on_last=False, ) assert_breached(evaluate(tmp_path), RULE) def test_missing_budget_in_the_snapshot_is_undetermined(tmp_path: Path) -> None: """规矩要的上限不在参数快照里就照实说缺什么,不硬编一个默认值。""" materialize( tmp_path, steps=2, stop_reason=StopReason.ACTION_BUDGET, max_actions=2, complete_on_last=False, ) def drop(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "run_started": del item["parameter_snapshot"]["request.max_actions"] return items edit_log(tmp_path / "soak-0001.jsonl", drop) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, RULE) is Verdict.UNDETERMINED rule = next(item for item in scoreboard.invariants if item.name == RULE) assert any("request.max_actions" in e.describe() for e in rule.undetermined) def test_every_stop_reason_has_a_rule(tmp_path: Path) -> None: """十个取值一个都不许落在「没有规矩」那条兜底路径上。 兜底路径本身留着,是给将来给 StopReason 加取值的人:那时它显式地报无法判定, 而不是静默地给一条绿。 """ del tmp_path covered = set(_STOP_REASON_RULES) | {StopReason.TASK_COMPLETED} assert covered == set(StopReason) def test_cancelled_with_run_finished_passes(tmp_path: Path) -> None: materialize(tmp_path, stop_reason=StopReason.CANCELLED, complete_on_last=False) assert verdict_of(evaluate(tmp_path), "停止原因与轨迹自洽") is Verdict.PASSED # --------------------------------------------------------------------------- # 缺文件:无法判定,既不是通过也不是击穿 # --------------------------------------------------------------------------- def test_missing_result_file_is_undetermined(tmp_path: Path) -> None: materialize(tmp_path, write_result=False) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "跨进程的结果与内存里的一致") is Verdict.UNDETERMINED assert scoreboard.verdict is Verdict.UNDETERMINED assert any(".result.json" in note for note in scoreboard.notes) def test_missing_meta_file_is_undetermined(tmp_path: Path) -> None: materialize(tmp_path, write_meta=False) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "事件条数等于本进程走完的步数") is Verdict.UNDETERMINED assert verdict_of(scoreboard, "投递失败计数对得上") is Verdict.UNDETERMINED assert not scoreboard.breached def test_missing_events_file_is_undetermined(tmp_path: Path) -> None: materialize(tmp_path, write_events=False) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "事件条数等于本进程走完的步数") is Verdict.UNDETERMINED assert verdict_of(scoreboard, "不串台") is Verdict.UNDETERMINED assert not scoreboard.breached def test_sigkilled_run_keeps_only_the_log(tmp_path: Path) -> None: """进程被 SIGKILL:只剩日志,而且尾行撕裂。整批判成无法判定,不是通过也不是击穿。""" materialize(tmp_path, write_result=False, write_events=False, write_meta=False) path = tmp_path / "soak-0001.jsonl" lines = path.read_text(encoding="utf-8").splitlines()[:-1] path.write_text("\n".join(lines) + '\n{"record": "run_fin', encoding="utf-8") scoreboard = evaluate(tmp_path) assert not scoreboard.breached assert scoreboard.verdict is Verdict.UNDETERMINED def materialize_stepless(runs_dir: Path, run_id: str = "soak-0001") -> None: """造一个零步的 run:只有 run_started 与 run_finished,没有意图、没有步、没有事件。 停止原因取 `cancelled`,因为十个原因里只有它的规矩不约束轨迹——它只要求日志里有结束 记录,而这个夹具本来就有。换成别的会顺带撞出那条规矩的击穿(`task_completed` 撞 「零步不可能完成」,`llm_error` 撞「至少有一步」),把这里要看的东西盖住。 """ materialize( runs_dir, run_id, steps=0, stop_reason=StopReason.CANCELLED, complete_on_last=False, ) def test_empty_log_cannot_be_judged_readable(tmp_path: Path) -> None: """一条被换行终结的行都没有:没有任何一行被读回来过,说「读得回来」没有依据。 造的是真实形态:`write_run_started` 先建文件、再写那一行,杀在两者之间就只剩一个空 文件,另外三个文件根本来不及写。 """ materialize(tmp_path, write_result=False, write_events=False, write_meta=False) (tmp_path / "soak-0001.jsonl").write_text("", encoding="utf-8") scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "日志能被读回来") is Verdict.UNDETERMINED assert not scoreboard.breached def test_log_with_only_a_torn_half_line_cannot_be_judged_readable(tmp_path: Path) -> None: materialize(tmp_path) (tmp_path / "soak-0001.jsonl").write_text('{"record": "run_star', encoding="utf-8") scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "日志能被读回来") is Verdict.UNDETERMINED assert any("撕裂尾行" in note for note in scoreboard.notes) def test_one_line_is_enough_to_judge_readability(tmp_path: Path) -> None: """下限是一行,别为了整齐往上抬:一行就足以判它解不解得开。""" materialize(tmp_path) path = tmp_path / "soak-0001.jsonl" first = path.read_text(encoding="utf-8").splitlines()[0] path.write_text(first + "\n", encoding="utf-8") assert verdict_of(evaluate(tmp_path), "日志能被读回来") is Verdict.PASSED path.write_text("{这一行解不开\n", encoding="utf-8") assert_breached(evaluate(tmp_path), "日志能被读回来") def test_run_without_intents_cannot_judge_their_homes(tmp_path: Path) -> None: materialize_stepless(tmp_path) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "意图都有归宿") is Verdict.UNDETERMINED assert not scoreboard.breached def test_intents_without_any_result_records_are_still_judged(tmp_path: Path) -> None: """有意图、没有任何结果记录,正是这条要判的那种,不许赖成判不了。""" materialize(tmp_path, steps=2) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [item for item in items if item["record"] in {"run_started", "intent"}], ) assert_breached(evaluate(tmp_path), "意图都有归宿") def test_one_intent_is_enough_to_judge_its_home(tmp_path: Path) -> None: """下限是一条意图:一条就足以判它悬不悬空、以及悬空的是不是最后一条。""" materialize(tmp_path, steps=1, stop_reason=StopReason.CANCELLED, complete_on_last=False) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [items[0], items[1]], ) # 唯一那条意图悬空,而它就是最后一条——那是崩溃点,判通过,不是判不了。 assert verdict_of(evaluate(tmp_path), "意图都有归宿") is Verdict.PASSED def test_run_without_step_payloads_cannot_judge_pairing(tmp_path: Path) -> None: materialize_stepless(tmp_path) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "步记录的内部不变量") is Verdict.UNDETERMINED assert not scoreboard.breached def test_pairing_counts_tagged_lines_not_decoded_records(tmp_path: Path) -> None: """下限数的是打着标签的行,不是解出来的记录。 唯一那条 `step_completed` 因为违反配对而解不出记录——按记录数当下限的话这条会报「判不 了」,可它要判的对象恰恰就是这一行。 """ materialize(tmp_path, steps=1, stop_reason=StopReason.CANCELLED, complete_on_last=False) def unpair(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed": item["result_id"] = None return items edit_log(tmp_path / "soak-0001.jsonl", unpair) assert_breached(evaluate(tmp_path), "步记录的内部不变量") def test_run_without_step_records_cannot_judge_agreement(tmp_path: Path) -> None: materialize_stepless(tmp_path) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "动作结果与步记录一致") is Verdict.UNDETERMINED assert not scoreboard.breached def test_step_without_an_action_outcome_still_counts_as_data(tmp_path: Path) -> None: """下限是一条步记录,不要求它带动作结果。 没有动作结果的那一档也在这条的判定范围里——那时步记录的 `action_status` 必须为空。 所以一份全是解析失败的日志确实验到了这条的一部分,报判不了反而是假的。 """ materialize(tmp_path, steps=1, stop_reason=StopReason.CANCELLED, complete_on_last=False) def strip_outcome(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed": item["result_id"] = None item["action_outcome"] = None item["step"]["action_status"] = None return items edit_log(tmp_path / "soak-0001.jsonl", strip_outcome) assert verdict_of(evaluate(tmp_path), "动作结果与步记录一致") is Verdict.PASSED def relabel(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed": item["step"]["action_status"] = "executed" return items edit_log(tmp_path / "soak-0001.jsonl", relabel) assert_breached(evaluate(tmp_path), "动作结果与步记录一致") def test_empty_events_file_leaves_the_event_half_unjudged(tmp_path: Path) -> None: """两半各判各的:日志那半判过了,也不能替事件那半的真空背书。""" materialize_stepless(tmp_path) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "不串台") is Verdict.UNDETERMINED assert not scoreboard.breached # 日志那半仍然是真判的:改掉一条记录的 run_id 照样击穿。 edit_log( tmp_path / "soak-0001.jsonl", lambda items: [ {**item, "run_id": "soak-9999"} if item["record"] == "run_finished" else item for item in items ], ) assert_breached(evaluate(tmp_path), "不串台") def test_log_without_records_leaves_the_log_half_unjudged(tmp_path: Path) -> None: """事件那半判过了(两条事件的 run_id 都对),也不能替日志那半的真空背书。""" materialize(tmp_path, steps=2) (tmp_path / "soak-0001.jsonl").write_text("", encoding="utf-8") scoreboard = evaluate(tmp_path) crosstalk = next(item for item in scoreboard.invariants if item.name == "不串台") assert crosstalk.verdict is Verdict.UNDETERMINED assert crosstalk.breaches == () def test_run_without_steps_cannot_judge_step_indices(tmp_path: Path) -> None: """零步的 run 上「步号从 0 开始逐 1 递增」根本没被验过,所以不许报通过。 崩溃注入那两类产物里真的会出现零步的 run——进程在第一步落盘之前就被杀了。 """ materialize( tmp_path, steps=0, stop_reason=StopReason.CANCELLED, complete_on_last=False, ) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "步号连续") is Verdict.UNDETERMINED assert not scoreboard.breached @pytest.mark.parametrize("steps", [0, 1]) def test_too_few_steps_cannot_judge_prompt_monotonicity(tmp_path: Path, steps: int) -> None: """零步和一步都凑不出相邻的两个值,一次比较都没发生过。""" materialize( tmp_path, steps=steps, stop_reason=StopReason.CANCELLED, complete_on_last=False, ) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "提示词字符数单调不减") is Verdict.UNDETERMINED assert not scoreboard.breached def test_one_step_still_judges_the_step_index(tmp_path: Path) -> None: """一步凑不出单调性,但「从 0 开始」验得了——两条的数据下限不一样。""" materialize(tmp_path, steps=1, stop_reason=StopReason.CANCELLED, complete_on_last=False) assert verdict_of(evaluate(tmp_path), "步号连续") is Verdict.PASSED def shift(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed": item["step"]["step_idx"] = 3 return items edit_log(tmp_path / "soak-0001.jsonl", shift) assert_breached(evaluate(tmp_path), "步号连续") def test_two_steps_are_enough_for_both(tmp_path: Path) -> None: """数据够了就必须真的判,不许赖着报无法判定。""" materialize(tmp_path, steps=2) scoreboard = evaluate(tmp_path) assert verdict_of(scoreboard, "步号连续") is Verdict.PASSED assert verdict_of(scoreboard, "提示词字符数单调不减") is Verdict.PASSED def test_empty_directory_is_undetermined_not_passed(tmp_path: Path) -> None: scoreboard = evaluate(tmp_path) assert scoreboard.verdict is Verdict.UNDETERMINED assert all(item.verdict is Verdict.UNDETERMINED for item in scoreboard.invariants) def test_events_file_is_not_mistaken_for_a_run(tmp_path: Path) -> None: materialize(tmp_path) scoreboard = evaluate(tmp_path) assert [item.run_id for item in scoreboard.summaries] == ["soak-0001"] # --------------------------------------------------------------------------- # 报告 # --------------------------------------------------------------------------- def _table_rows(report: str) -> list[list[str]]: return [re.split(r"(? None: materialize(tmp_path, "soak-0001", steps=2) materialize( tmp_path, "soak-0002", steps=4, stop_reason=StopReason.STEP_BUDGET, max_steps=4, complete_on_last=False, ) report = render_report(evaluate(tmp_path)) assert report.startswith("# PolyLoop 压测记分板") for heading in ( "## 总判定", "## 不变量", "## 观察", "## 统计", "## 与基线的对照", "## 逐个 run", ): assert heading in report # 表格的每一行都以竖线开头结尾,且同一段里的列数一致。 widths: set[int] = set() previous_was_table = False for line in report.splitlines(): if line.startswith("|"): assert line.endswith("|"), line if not previous_was_table: widths = set() widths.add(len(re.split(r"(? None: materialize(tmp_path, steps=2) report = render_report(evaluate(tmp_path)) assert "937" in report assert "915" in report # 基线的 task_completed 计数 assert "步数 p90" in report assert "对照不是判定" in report def test_report_never_contains_document_fragments(tmp_path: Path) -> None: """观察与最终回答里的哨兵一个字都不许进报告——压测语料里有第三方的真实文档。""" result = materialize( tmp_path, steps=2, observation=SENTINEL, stop_reason=StopReason.AGENT_FINISHED, complete_on_last=False, final_answer=SENTINEL, ) # 再造一处击穿,让证据渲染那条路径也被走到。 payload = encode(result) payload["final_answer"] = SENTINEL + "-改过的" payload["steps"][0]["observation"] = SENTINEL + "-也改过" # type: ignore[index] payload["steps"][0]["tool_name"] = SENTINEL + "-当成工具名" # type: ignore[index] payload["steps"][1]["tool_arguments"] = SENTINEL + "-当成参数" # type: ignore[index] (tmp_path / "soak-0001.result.json").write_text( json.dumps(payload, ensure_ascii=False), encoding="utf-8" ) scoreboard = evaluate(tmp_path) assert scoreboard.breached report = render_report(scoreboard) assert SENTINEL not in report assert "SENTINEL" not in report def test_report_does_not_echo_a_hostile_record_tag(tmp_path: Path) -> None: materialize(tmp_path) edit_log( tmp_path / "soak-0001.jsonl", lambda items: [{**items[0], "record": SENTINEL}] + items[1:], ) scoreboard = evaluate(tmp_path) assert_breached(scoreboard, "日志能被读回来") report = render_report(scoreboard) assert SENTINEL not in report assert "SENTINEL" not in report def test_report_does_not_echo_a_hostile_observation_through_decode_errors( tmp_path: Path, ) -> None: """解码失败的消息也不许把观察带出来。库的取值函数对字符串字段只报类型不报内容。""" materialize(tmp_path, observation=SENTINEL) def corrupt(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed": # 少一个必填字段:整行解不出来,消息里只会有字段名。 del item["step"]["prompt_chars"] return items edit_log(tmp_path / "soak-0001.jsonl", corrupt) scoreboard = evaluate(tmp_path) assert_breached(scoreboard, "日志能被读回来") report = render_report(scoreboard) assert SENTINEL not in report assert "SENTINEL" not in report def test_report_keeps_tool_names_out(tmp_path: Path) -> None: """工具名也不进报告。 一次没通过校验的工具调用会把模型编的那串原样记进 `tool_name`,所以它和观察同档, 不是「我们自己的枚举取值」。这条被一次实测撞出来过:先前的实现把工具名过一遍字符 白名单就放行,而白名单留下的正是 ASCII 那一段,哨兵原样穿了过去。 """ materialize( tmp_path, complete_on_last=False, tool_name="工具|名里有竖线和" + SENTINEL, ) scoreboard = evaluate(tmp_path, completing_tools=frozenset({"finish"})) assert_breached(scoreboard, "停止原因与轨迹自洽") report = render_report(scoreboard) assert SENTINEL not in report assert "SENTINEL" not in report assert "竖线" not in report # --------------------------------------------------------------------------- # 命令行与退出码 # --------------------------------------------------------------------------- def test_main_returns_zero_on_a_clean_batch(tmp_path: Path) -> None: runs = tmp_path / "runs" report = tmp_path / "reports" / "batch.md" materialize(runs) assert main(["--runs-dir", str(runs), "--report", str(report)]) == 0 assert report.read_text(encoding="utf-8").startswith("# PolyLoop 压测记分板") def test_main_returns_nonzero_on_a_breach(tmp_path: Path) -> None: runs = tmp_path / "runs" materialize(runs, steps=3, events=1) code = main(["--runs-dir", str(runs), "--report", str(tmp_path / "r.md")]) assert code == EXIT_BREACHED def test_main_returns_nonzero_on_undetermined(tmp_path: Path) -> None: runs = tmp_path / "runs" materialize(runs, write_meta=False) argv = ["--runs-dir", str(runs), "--report", str(tmp_path / "r.md")] assert main(argv) == EXIT_UNDETERMINED assert main([*argv, "--allow-undetermined"]) == 0 def test_allow_undetermined_does_not_forgive_a_breach(tmp_path: Path) -> None: runs = tmp_path / "runs" materialize(runs, steps=3, write_meta=False) def bump(items: list[dict]) -> list[dict]: for item in items: if item["record"] == "step_completed" and item["step"]["step_idx"] == 1: item["step"]["step_idx"] = 2 return items edit_log(runs / "soak-0001.jsonl", bump) code = main( [ "--runs-dir", str(runs), "--report", str(tmp_path / "r.md"), "--allow-undetermined", ] ) assert code == EXIT_BREACHED def test_completing_tool_flag_reaches_the_invariant(tmp_path: Path) -> None: runs = tmp_path / "runs" materialize(runs, complete_on_last=False, tool_name="finish_task") argv = ["--runs-dir", str(runs), "--report", str(tmp_path / "r.md")] assert main(argv) == EXIT_UNDETERMINED assert main([*argv, "--completing-tool", "finish_task"]) == 0 def test_baseline_flags_override_the_defaults(tmp_path: Path) -> None: runs = tmp_path / "runs" report = tmp_path / "r.md" materialize(runs) code = main( [ "--runs-dir", str(runs), "--report", str(report), "--baseline-label", "自造基线", "--baseline-stop-reason", "task_completed=7", "--baseline-steps-p50", "3", ] ) assert code == 0 text = report.read_text(encoding="utf-8") assert "自造基线" in text assert "915" not in text def test_bad_baseline_pair_is_rejected(tmp_path: Path) -> None: with pytest.raises(SystemExit): main( [ "--runs-dir", str(tmp_path), "--report", str(tmp_path / "r.md"), "--baseline-stop-reason", "没有等号", ] )