feat(soak): 补上 context_overflow 与 env_error 两类,七类变九类

一次 193 个运行的全量跑完之后,停止原因的十个取值里有两个一次都没出现过。没出现不等于
它们是对的,只等于没验过——这正是「全绿要先怀疑负载」该指向的地方。

context_overflow 的判据不能照字面写成「最后一步的提示词超过上限」:规模判定在调模型之前
做,命中时不产生步记录,所以落盘的每条步记录必定不超上限,那样断言等于断言契约的反面。
改成判「再走一步会有多大」,公式拿全量里 865 对相邻步验过,0 处不符。

env_error 是真把容器 docker kill 掉,不是用测试替身。它自己起一个池、用另一个端口——
共用那个 size=1 的池的话,排在它后面的每一类都会跑在一个不存在的环境上。

实跑:两类都通过,击穿 0、无法判定 0。

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-08-11 11:14:04 -04:00
parent 5d5371792d
commit 67f0d355d0
2 changed files with 1330 additions and 61 deletions
+779 -56
View File
File diff suppressed because it is too large Load Diff
+551 -5
View File
@@ -21,57 +21,81 @@ import json
import sys import sys
from pathlib import Path from pathlib import Path
import httpx
import pytest import pytest
from polyloop.ports import EventKind, InvalidDecision, RunLog from polyloop.ports import EventKind, InvalidDecision, RunLog
from polyloop.session import RunRequest
from polyloop.tools import ToolRegistry
from polyloop.types import ( from polyloop.types import (
ActionOutcome, ActionOutcome,
ActionStatus, ActionStatus,
Budget,
Context,
Injection,
Intent, Intent,
IntentKind, IntentKind,
Message,
ModelCallResult, ModelCallResult,
ModelReply, ModelReply,
ReplayPolicy, ReplayPolicy,
Role,
RunFinished, RunFinished,
RunResult, RunResult,
RunStarted, RunStarted,
StepCompleted, StepCompleted,
StepRecord, StepRecord,
StopReason, StopReason,
TextBlock,
) )
from tools.soak.faults import ( from tools.soak.faults import (
CONTEXT_OVERFLOW_SLACK_CHARS,
CRASH_EXIT_CODE, CRASH_EXIT_CODE,
FAULT_NAMES,
AlwaysInvalidParser, AlwaysInvalidParser,
CallGuard, CallGuard,
Criterion, Criterion,
CriterionStatus, CriterionStatus,
EnvBreakingExecutor,
FaultReport, FaultReport,
JsonlEventSink, JsonlEventSink,
KillTiming, KillTiming,
LogRead, LogRead,
SelfKillingStore, SelfKillingStore,
build_context_overflow_budget,
build_parser, build_parser,
check_all_steps_parse_failed, check_all_steps_parse_failed,
check_at_least_one_step,
check_audit_unchanged, check_audit_unchanged,
check_cancelled_raised, check_cancelled_raised,
check_crash_prefix_preserved, check_crash_prefix_preserved,
check_env_broken_after_a_full_step,
check_env_untouched, check_env_untouched,
check_executed_action_count, check_executed_action_count,
check_intents_settled, check_intents_settled,
check_last_observation_is_synthetic,
check_last_step_action_status,
check_lease_returned, check_lease_returned,
check_log_readable, check_log_readable,
check_never_action_not_replayed, check_never_action_not_replayed,
check_no_env_error_step, check_no_env_error_step,
check_prompt_chars_monotonic,
check_prompt_reached_max_prompt_chars,
check_resume_made_progress, check_resume_made_progress,
check_run_finished_present,
check_step_count, check_step_count,
check_step_indices_dense, check_step_indices_dense,
check_steps_before_last_all_executed,
check_stop_reason, check_stop_reason,
container_name_for_port,
count_model_calls, count_model_calls,
guarded, guarded,
initial_prompt_chars,
main, main,
parse_audit_line, parse_audit_line,
parse_terminated, parse_terminated,
read_log, read_log,
should_break_env,
should_kill, should_kill,
spawn_and_kill, spawn_and_kill,
stop_reason_of, stop_reason_of,
@@ -79,6 +103,7 @@ from tools.soak.faults import (
terminated_prefix, terminated_prefix,
write_sidecars, write_sidecars,
) )
from tools.soak.scenarios.appworld import build_synthetic_observations
from tools.soak.scenarios.govdoc import AUDIT_LOG_NAME from tools.soak.scenarios.govdoc import AUDIT_LOG_NAME
RUN_ID = "fault-test-0" RUN_ID = "fault-test-0"
@@ -123,6 +148,10 @@ def step_completed(
status: str | None = "executed", status: str | None = "executed",
parse_ok: bool = True, parse_ok: bool = True,
action_status: str | None = "executed", action_status: str | None = "executed",
prompt_chars: object = 10,
raw_output: str = "x",
observation: str = "o",
observation_is_synthetic: bool = False,
) -> dict[str, object]: ) -> dict[str, object]:
outcome = ( outcome = (
None None
@@ -142,16 +171,16 @@ def step_completed(
"action_outcome": outcome, "action_outcome": outcome,
"step": { "step": {
"step_idx": step_idx, "step_idx": step_idx,
"raw_output": "x", "raw_output": raw_output,
"content_chars": 1, "content_chars": len(raw_output),
"thinking_chars": 0, "thinking_chars": 0,
"action": None, "action": None,
"parse_ok": parse_ok, "parse_ok": parse_ok,
"parse_error": None if parse_ok else "解释不了", "parse_error": None if parse_ok else "解释不了",
"observation": "o", "observation": observation,
"observation_is_synthetic": False, "observation_is_synthetic": observation_is_synthetic,
"observation_truncated_chars": 0, "observation_truncated_chars": 0,
"prompt_chars": 10, "prompt_chars": prompt_chars,
"call_id": "c0", "call_id": "c0",
"step_wall_ms": 1, "step_wall_ms": 1,
"tool_name": None, "tool_name": None,
@@ -163,6 +192,29 @@ def step_completed(
} }
#: 造日志时用的观察模板。套一次观察多出 `len("观察:\n") == 4` 个字符,判据算「再走一步的提示词
#: 会有多大」时要把这四个字符算进去。
TEMPLATE = "观察:{observation}\n"
def run_started(
*, max_prompt_chars: int = 100, observation_template: str = TEMPLATE
) -> dict[str, object]:
"""运行开始那条记录。参数快照里只放判据会读的两个键。
真的快照还有十几个键,多放几个不会让任何一条判据的行为改变——它们按键名取值。
"""
return {
"record": "run_started",
"run_id": RUN_ID,
"parameter_snapshot": {
"request.max_prompt_chars": str(max_prompt_chars),
"request.observation_template": observation_template,
},
"schema_version": 1,
}
def run_finished(*, stop_reason: str = "task_completed") -> dict[str, object]: def run_finished(*, stop_reason: str = "task_completed") -> dict[str, object]:
return { return {
"record": "run_finished", "record": "run_finished",
@@ -1181,3 +1233,497 @@ def test_parser_accepts_repeated_fault_flags() -> None:
] ]
) )
assert args.fault == ["cancel_model", "cancel_env"] assert args.fault == ["cancel_model", "cancel_env"]
# ---------------------------------------------------------------------------
# 十五、撞提示词上限
# ---------------------------------------------------------------------------
def test_run_finished_present_passes_and_breaches() -> None:
"""结束记录在不在与它带的取值对不对是两条判据,成因不同。"""
assert check_run_finished_present(as_read(run_finished())).status is CriterionStatus.PASSED
assert check_run_finished_present(as_read(step_completed())).status is CriterionStatus.BREACHED
def test_stop_reason_context_overflow_passes_and_breaches() -> None:
good = as_read(step_completed(), run_finished(stop_reason="context_overflow"))
bad = as_read(step_completed(), run_finished(stop_reason="step_budget"))
assert check_stop_reason(good, "context_overflow").status is CriterionStatus.PASSED
breach = check_stop_reason(bad, "context_overflow")
assert breach.status is CriterionStatus.BREACHED
assert "step_budget" in breach.evidence
def test_at_least_one_step_passes() -> None:
assert check_at_least_one_step(as_read(step_completed())).status is CriterionStatus.PASSED
def test_at_least_one_step_is_undetermined_without_steps() -> None:
"""第一步就撞上限的话这一类什么都没验到,报「无法判定」不报「击穿」。"""
outcome = check_at_least_one_step(as_read(run_started(), run_finished()))
assert outcome.status is CriterionStatus.UNDETERMINED
assert "一条步记录都没有" in outcome.evidence
def test_prompt_chars_monotonic_passes() -> None:
read = as_read(
step_completed(step_idx=0, prompt_chars=100),
step_completed(step_idx=1, prompt_chars=100),
step_completed(step_idx=2, prompt_chars=250),
)
outcome = check_prompt_chars_monotonic(read)
assert outcome.status is CriterionStatus.PASSED
assert "100" in outcome.evidence and "250" in outcome.evidence
def test_prompt_chars_monotonic_breaches_on_a_drop() -> None:
"""一次下降就是历史被截断过——那是这件事在轨迹里唯一看得见的痕迹。"""
read = as_read(
step_completed(step_idx=0, prompt_chars=100),
step_completed(step_idx=1, prompt_chars=300),
step_completed(step_idx=2, prompt_chars=120),
)
outcome = check_prompt_chars_monotonic(read)
assert outcome.status is CriterionStatus.BREACHED
assert "第 2 步从 300 掉到 120" in outcome.evidence
def test_prompt_chars_monotonic_breaches_on_a_non_integer() -> None:
read = as_read(step_completed(prompt_chars="很多"))
assert check_prompt_chars_monotonic(read).status is CriterionStatus.BREACHED
def test_prompt_chars_monotonic_is_undetermined_without_steps() -> None:
assert check_prompt_chars_monotonic(as_read(run_started())).status is (
CriterionStatus.UNDETERMINED
)
def test_prompt_reached_max_prompt_chars_passes() -> None:
"""最后一步自己装得下,再走一步就装不下——那才是「撑到了上限那条线上」。
上限 100;最后一步的提示词 90,它的输出 5 个字符、观察 10 个字符再加模板的 4 个,
下一次装配是 109 字符。
"""
read = as_read(
run_started(max_prompt_chars=100),
step_completed(prompt_chars=90, raw_output="x" * 5, observation="y" * 10),
run_finished(stop_reason="context_overflow"),
)
outcome = check_prompt_reached_max_prompt_chars(read)
assert outcome.status is CriterionStatus.PASSED
assert "109" in outcome.evidence
def test_prompt_reached_max_prompt_chars_breaches_when_it_still_fits() -> None:
"""还装得下就报 context_overflow:这次运行根本没撞到上限。"""
read = as_read(
run_started(max_prompt_chars=100),
step_completed(prompt_chars=50, raw_output="x", observation="y"),
run_finished(stop_reason="context_overflow"),
)
outcome = check_prompt_reached_max_prompt_chars(read)
assert outcome.status is CriterionStatus.BREACHED
assert "仍不超过上限 100" in outcome.evidence
def test_prompt_reached_max_prompt_chars_breaches_when_an_oversized_prompt_was_admitted() -> None:
"""步记录自己的提示词就超了上限,说明有一次超限的装配被放行去调了模型。"""
read = as_read(
run_started(max_prompt_chars=100),
step_completed(prompt_chars=120, raw_output="x", observation="y"),
run_finished(stop_reason="context_overflow"),
)
outcome = check_prompt_reached_max_prompt_chars(read)
assert outcome.status is CriterionStatus.BREACHED
assert "被放行去调了模型" in outcome.evidence
def test_prompt_reached_max_prompt_chars_reads_the_limit_from_the_snapshot() -> None:
"""上限从参数快照读,不硬编码:同一条步记录换个上限就该翻面。"""
step = step_completed(prompt_chars=90, raw_output="x" * 5, observation="y" * 10)
tight = as_read(run_started(max_prompt_chars=100), step)
loose = as_read(run_started(max_prompt_chars=100_000), step)
assert check_prompt_reached_max_prompt_chars(tight).status is CriterionStatus.PASSED
assert check_prompt_reached_max_prompt_chars(loose).status is CriterionStatus.BREACHED
def test_prompt_reached_max_prompt_chars_is_undetermined_without_run_started() -> None:
read = as_read(step_completed(), run_finished(stop_reason="context_overflow"))
outcome = check_prompt_reached_max_prompt_chars(read)
assert outcome.status is CriterionStatus.UNDETERMINED
assert "run_started" in outcome.evidence
def test_prompt_reached_max_prompt_chars_is_undetermined_without_steps() -> None:
read = as_read(run_started(), run_finished(stop_reason="context_overflow"))
assert check_prompt_reached_max_prompt_chars(read).status is CriterionStatus.UNDETERMINED
def test_prompt_reached_max_prompt_chars_is_undetermined_on_a_bad_snapshot() -> None:
read = as_read(
{
"record": "run_started",
"run_id": RUN_ID,
"parameter_snapshot": {
"request.max_prompt_chars": "很多",
"request.observation_template": TEMPLATE,
},
"schema_version": 1,
},
step_completed(),
)
assert check_prompt_reached_max_prompt_chars(read).status is CriterionStatus.UNDETERMINED
# ---------------------------------------------------------------------------
# 十六、按上下文现算提示词上限
# ---------------------------------------------------------------------------
def a_request(
*,
run_level: tuple[str, ...] = ("系统段",),
goal_level: tuple[str, ...] = ("题面",),
injections: dict[str, tuple[Injection, ...]] | None = None,
) -> RunRequest:
"""一份只填了装配用得着那几样的请求。执行器是个不派生自注册表的普通对象,构造期不比对。"""
class _Executor:
def parameters(self) -> dict[str, str]:
return {"kind": "fake"}
async def execute(self, action: object) -> ActionOutcome: # pragma: no cover - 用不到
raise AssertionError("这份请求只用来算提示词规模")
return RunRequest(
run_id=RUN_ID,
budget=Budget(
max_steps=1, max_actions=1, max_consecutive_parse_failures=1, max_prompt_chars=1
),
action_executor=_Executor(), # type: ignore[arg-type]
tools=ToolRegistry(),
context=Context(
run_level=tuple(
Message(role=Role.SYSTEM, content=(TextBlock(text=text),)) for text in run_level
),
goal_level=tuple(
Message(role=Role.USER, content=(TextBlock(text=text),)) for text in goal_level
),
),
injections=injections or {},
model_binding={},
model_replay_policy=ReplayPolicy.NEVER,
observation_template=TEMPLATE,
cancel_grace_seconds=1.0,
)
def test_initial_prompt_chars_counts_every_segment() -> None:
request = a_request(
run_level=("a" * 10, "b" * 5),
goal_level=("c" * 7,),
injections={"skill": (Injection(entry_id="e0", content="d" * 3),)},
)
assert initial_prompt_chars(request) == 25
def test_context_overflow_budget_leaves_room_for_exactly_the_first_step() -> None:
"""上限 = 初始提示词 + 余量。第一步刚好装得下,第二步靠一步的增长撑过去。"""
request = a_request(run_level=("x" * 40,), goal_level=("y" * 60,))
budget = build_context_overflow_budget(request)
assert budget.max_prompt_chars == 100 + CONTEXT_OVERFLOW_SLACK_CHARS
def test_initial_prompt_chars_rejects_an_unknown_block_type() -> None:
"""认不得的块当成 0 会让上限算小,于是第一步就撞上限、这一类什么都验不到。"""
class _Weird:
pass
request = a_request()
broken = replace_context_block(request, _Weird())
with pytest.raises(Exception, match="认不得的内容块类型"):
initial_prompt_chars(broken)
def replace_context_block(request: RunRequest, block: object) -> RunRequest:
"""把上下文里那条消息的内容块换成给定的东西。造非法输入用。"""
import dataclasses
context = Context(
run_level=(Message(role=Role.SYSTEM, content=(block,)),), # type: ignore[arg-type]
goal_level=(),
)
return dataclasses.replace(request, context=context)
# ---------------------------------------------------------------------------
# 十七、环境故障:判据
# ---------------------------------------------------------------------------
def test_stop_reason_env_error_passes_and_breaches() -> None:
good = as_read(step_completed(), run_finished(stop_reason="env_error"))
bad = as_read(step_completed(), run_finished(stop_reason="task_completed"))
assert check_stop_reason(good, "env_error").status is CriterionStatus.PASSED
assert check_stop_reason(bad, "env_error").status is CriterionStatus.BREACHED
def test_last_step_action_status_passes() -> None:
read = as_read(
step_completed(step_idx=0, action_status="executed"),
step_completed(step_idx=1, action_status="env_error"),
)
assert check_last_step_action_status(read, expected="env_error").status is (
CriterionStatus.PASSED
)
def test_last_step_action_status_breaches() -> None:
read = as_read(
step_completed(step_idx=0, action_status="env_error"),
step_completed(step_idx=1, action_status="executed"),
)
outcome = check_last_step_action_status(read, expected="env_error")
assert outcome.status is CriterionStatus.BREACHED
assert "executed" in outcome.evidence
def test_last_step_action_status_is_undetermined_without_steps() -> None:
assert check_last_step_action_status(as_read(run_started()), expected="env_error").status is (
CriterionStatus.UNDETERMINED
)
def test_env_failed_observation_passes_with_the_value_from_the_scenario() -> None:
"""期望值从场景那份合成观察取,判据这边不抄一份字面量。"""
expected = build_synthetic_observations().env_failed
read = as_read(
step_completed(
action_status="env_error", observation=expected, observation_is_synthetic=True
)
)
assert check_last_observation_is_synthetic(read, expected=expected, name="x").status is (
CriterionStatus.PASSED
)
def test_env_failed_observation_breaches_when_the_flag_is_false() -> None:
expected = build_synthetic_observations().env_failed
read = as_read(
step_completed(
action_status="env_error", observation=expected, observation_is_synthetic=False
)
)
outcome = check_last_observation_is_synthetic(read, expected=expected, name="x")
assert outcome.status is CriterionStatus.BREACHED
assert "observation_is_synthetic" in outcome.evidence
def test_env_failed_observation_breaches_when_the_executor_text_survived() -> None:
"""替换没发生的话,历史里躺着的是与环境通信那一层给的原文——那正是这一条要抓的。"""
expected = build_synthetic_observations().env_failed
read = as_read(
step_completed(
action_status="env_error",
observation="ConnectError: All connection attempts failed",
observation_is_synthetic=True,
)
)
outcome = check_last_observation_is_synthetic(read, expected=expected, name="x")
assert outcome.status is CriterionStatus.BREACHED
assert "ConnectError" not in outcome.evidence
def test_env_failed_observation_is_undetermined_without_steps() -> None:
assert (
check_last_observation_is_synthetic(as_read(run_started()), expected="x", name="x").status
is CriterionStatus.UNDETERMINED
)
def test_steps_before_last_all_executed_passes() -> None:
read = as_read(
step_completed(step_idx=0, action_status="executed"),
step_completed(step_idx=1, action_status="executed"),
step_completed(step_idx=2, action_status="env_error"),
)
assert check_steps_before_last_all_executed(read).status is CriterionStatus.PASSED
def test_steps_before_last_all_executed_breaches() -> None:
"""环境坏掉之前的步被改写或补上别的状态,说明一次局部故障扩散到了已经落地的轨迹上。"""
read = as_read(
step_completed(step_idx=0, action_status="executed"),
step_completed(step_idx=1, action_status="env_error"),
step_completed(step_idx=2, action_status="env_error"),
)
outcome = check_steps_before_last_all_executed(read)
assert outcome.status is CriterionStatus.BREACHED
assert "[1]" in outcome.evidence
def test_steps_before_last_all_executed_is_undetermined_with_a_single_step() -> None:
"""只有一步的话这一条真空成立,那种「通过」什么都没验。"""
outcome = check_steps_before_last_all_executed(as_read(step_completed()))
assert outcome.status is CriterionStatus.UNDETERMINED
assert "真空成立" in outcome.evidence
def test_env_broken_after_a_full_step_passes() -> None:
outcome = check_env_broken_after_a_full_step(broken=True, executions_before_break=1)
assert outcome.status is CriterionStatus.PASSED
def test_env_broken_after_a_full_step_is_undetermined_when_it_never_broke() -> None:
outcome = check_env_broken_after_a_full_step(broken=False, executions_before_break=0)
assert outcome.status is CriterionStatus.UNDETERMINED
assert "一次都没被弄坏" in outcome.evidence
def test_env_broken_after_a_full_step_is_undetermined_when_it_broke_too_early() -> None:
"""第一次执行之前就打死容器的话,压到的是初始化而不是动作执行接缝。"""
outcome = check_env_broken_after_a_full_step(broken=True, executions_before_break=0)
assert outcome.status is CriterionStatus.UNDETERMINED
assert "初始化" in outcome.evidence
# ---------------------------------------------------------------------------
# 十八、环境故障:杀容器那段编排里的纯函数与执行器包装
# ---------------------------------------------------------------------------
def test_container_name_for_port() -> None:
assert container_name_for_port(8201) == "polyloop-soak-appworld-8201"
def test_should_break_env_waits_for_a_full_execution() -> None:
assert (
should_break_env(executions_done=0, break_after_executions=1, already_broken=False) is False
)
assert (
should_break_env(executions_done=1, break_after_executions=1, already_broken=False) is True
)
def test_should_break_env_never_breaks_twice() -> None:
"""容器已经没了,再发一次 docker kill 只会拿到一个「没有这个容器」的错误。"""
assert (
should_break_env(executions_done=5, break_after_executions=1, already_broken=True) is False
)
class _FakeExecutor:
"""按剧本一次次返回结果或抛异常的假执行器。"""
def __init__(self, script: list[object]) -> None:
self.script = script
self.calls = 0
def parameters(self) -> dict[str, str]:
return {"kind": "fake_appworld"}
async def execute(self, action: object) -> ActionOutcome:
item = self.script[self.calls]
self.calls += 1
if isinstance(item, BaseException):
raise item
return item # type: ignore[return-value]
def an_outcome(status: ActionStatus = ActionStatus.EXECUTED) -> ActionOutcome:
return ActionOutcome(
status=status,
observation="环境返回",
observation_is_synthetic=False,
env_reported_completion=False,
observation_truncated_chars=0,
)
async def test_env_breaking_executor_breaks_after_one_full_execution() -> None:
"""第一次执行之前不动手,第二次执行之前动手。"""
broke_at: list[int] = []
inner = _FakeExecutor([an_outcome(), httpx.ConnectError("All connection attempts failed")])
async def break_env() -> str:
broke_at.append(inner.calls)
return "容器已被打死"
executor = EnvBreakingExecutor(inner=inner, break_after_executions=1, break_env=break_env)
first = await executor.execute(object())
assert first.status is ActionStatus.EXECUTED
assert broke_at == []
second = await executor.execute(object())
assert broke_at == [1]
assert executor.broken is True
assert executor.executions_before_break == 1
assert executor.break_note == "容器已被打死"
assert second.status is ActionStatus.ENV_ERROR
async def test_env_breaking_executor_translates_a_transport_error() -> None:
"""场景那侧只接 AppWorldError,容器没了时抛的是 httpx.ConnectError,不翻译就整个抛出去。"""
inner = _FakeExecutor([httpx.ConnectError("All connection attempts failed")])
async def break_env() -> str:
return "打死了"
executor = EnvBreakingExecutor(inner=inner, break_after_executions=0, break_env=break_env)
outcome = await executor.execute(object())
assert outcome.status is ActionStatus.ENV_ERROR
assert outcome.observation_is_synthetic is False
assert "ConnectError" in outcome.observation
async def test_env_breaking_executor_passes_an_inner_env_error_through() -> None:
"""内层自己判出环境故障时原样透传:场景那边哪天补上转换,这里不用跟着改。"""
inner = _FakeExecutor([an_outcome(ActionStatus.ENV_ERROR)])
async def break_env() -> str: # pragma: no cover - 这条用例不动手
raise AssertionError("不该动手")
executor = EnvBreakingExecutor(inner=inner, break_after_executions=99, break_env=break_env)
outcome = await executor.execute(object())
assert outcome.status is ActionStatus.ENV_ERROR
assert outcome.observation == "环境返回"
async def test_env_breaking_executor_reports_its_probe_in_parameters() -> None:
inner = _FakeExecutor([])
async def break_env() -> str: # pragma: no cover - 这条用例不执行动作
raise AssertionError("不该动手")
executor = EnvBreakingExecutor(inner=inner, break_after_executions=1, break_env=break_env)
assert executor.parameters() == {"kind": "fake_appworld", "env_break_probe": "docker_kill"}
# ---------------------------------------------------------------------------
# 十九、两类新故障接进命令行
# ---------------------------------------------------------------------------
def test_new_faults_are_in_the_default_selection() -> None:
"""不给 --fault 时按 FAULT_NAMES 全跑,两类都得在里面。"""
assert "context_overflow" in FAULT_NAMES
assert "env_error" in FAULT_NAMES
def test_parser_accepts_the_new_faults() -> None:
args = build_parser().parse_args(
[
"--runs-dir",
"x",
"--budget-calls",
"10",
"--fault",
"context_overflow",
"--fault",
"env_error",
]
)
assert args.fault == ["context_overflow", "env_error"]