fix(soak): 父子两侧的崩溃条件对齐,并给自杀留一个窗口
上一轮实跑里确定性自杀一次都没走到——父进程一看见日志尾部形态对上就发信号,而自杀条件 比它严一档(还要求审计账非空),于是外部信号永远抢先,自杀路径成了死代码,崩溃点又变回 碰运气。表现是崩得太早:动作还没执行过,最硬的那条判据没有实料可判。 两处对齐:父进程的命中条件也要求审计账非空(做成必传参数,给默认值等于让某个调用点静默 跳过这一条,而这正是这次出问题的方式);兜底 SIGKILL 之前先等两秒看子进程是不是自己以 137 退出。 实跑结果:两档都走确定性自杀路径,都崩在「写入执行过之后」,绝不重放那条判据在两档都有 实料——审计账 1 条、去重后仍 1 条,续跑前后也都是 1 条。整套七类击穿 0 条、无法判定 0 条。 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
+78
-14
@@ -622,6 +622,14 @@ def check_lease_returned(*, borrowed: bool, timeout_s: float, pool_size: int) ->
|
||||
#: 取 137 是照 128+9 那个惯例(外部 SIGKILL 的等价形态),让两条路径在日志里读起来是同一件事。
|
||||
CRASH_EXIT_CODE = 137
|
||||
|
||||
#: 条件对上之后留给子进程自杀的窗口,秒。窗口内它还没死,父进程才兜底发 SIGKILL。
|
||||
#:
|
||||
#: 两秒对「进程执行完 `os._exit` 系统调用」来说是天文数字——这个数不是在等一件慢事,是在给
|
||||
#: 两个进程的观察顺序留一点余量:父进程是从文件里看见那条记录的,而子进程要先从写入调用返回、
|
||||
#: 再走几行 Python 才到自杀那一句。取小了会偶发地抢在它前面,而抢赢的表现是「这次又是兜底
|
||||
#: 命中的」,不是一个错误。
|
||||
SELF_KILL_GRACE_S = 2.0
|
||||
|
||||
|
||||
class KillTiming(StrEnum):
|
||||
"""在哪个时机让子进程崩掉。两种时机的判据不同,不许混成一个用例。"""
|
||||
@@ -633,7 +641,9 @@ class KillTiming(StrEnum):
|
||||
AT_INTENT = "at_intent"
|
||||
|
||||
|
||||
def should_kill(read: LogRead, *, timing: KillTiming, after_steps: int) -> bool:
|
||||
def should_kill(
|
||||
read: LogRead, *, timing: KillTiming, after_steps: int, audit_is_not_empty: bool
|
||||
) -> bool:
|
||||
"""现在这份日志尾部是不是要等的那个时机。纯函数。
|
||||
|
||||
子进程自杀之后父进程拿它复核一遍尾部形态;外部 SIGKILL 那条兜底路径每次轮询也问它一句。
|
||||
@@ -642,7 +652,15 @@ def should_kill(read: LogRead, *, timing: KillTiming, after_steps: int) -> bool:
|
||||
更严:GovDoc 的 `read_document` 与 `grep_document` 声明的是 `safe`,悬在那种意图上续跑
|
||||
会重放动作接着跑,停止原因不是 `resume_state_unknown`。放宽这一条,判据就会时对时错,
|
||||
而错的那些次看起来只是「模型这次走了别的路」。
|
||||
|
||||
**`audit_is_not_empty` 必传,两种时机都要求它为真。** 它与 `SelfKillingStore` 的自杀条件
|
||||
是同一条,两边必须逐字对齐:父进程这侧要是松一档(只看日志尾部),它总会在子进程走到自杀
|
||||
那一行之前抢先发出信号,于是自杀路径成了永远走不到的死代码,而崩溃点落在哪儿又变回碰运气。
|
||||
实测就是这么发生的——两次崩溃全是外部信号命中的,其中一次崩在审计账还是空的时候,最硬那条
|
||||
判据只能真空成立。
|
||||
"""
|
||||
if not audit_is_not_empty:
|
||||
return False
|
||||
if len(tagged(read, "step_completed")) < after_steps:
|
||||
return False
|
||||
if not read.payloads:
|
||||
@@ -673,6 +691,9 @@ class SelfKillingStore:
|
||||
**自杀条件带上「审计账已经非空」**:要验的最硬那条判据是「声明绝不重放的动作没有被执行
|
||||
两次」,它数的是工作区审计账。崩在模型还没调过 `write_note` 的时候,账是空的,那条判据
|
||||
真空成立——报出来是「通过」,实际什么都没验。
|
||||
|
||||
父进程那侧的兜底条件(`should_kill`)与这里逐字对齐,包括审计账那一项。松一档它就会每次
|
||||
抢先,这个类成为死代码;那件事实测发生过一次,理由写在 `should_kill` 的 docstring 里。
|
||||
"""
|
||||
|
||||
__slots__ = ("_exit_now", "_inner", "_timing", "_workspace")
|
||||
@@ -749,7 +770,13 @@ class KillOutcome:
|
||||
|
||||
|
||||
def _judge_crash(
|
||||
*, log_path: Path, timing: KillTiming, after_steps: int, exit_code: int | None, how: str
|
||||
*,
|
||||
log_path: Path,
|
||||
workspace: Path,
|
||||
timing: KillTiming,
|
||||
after_steps: int,
|
||||
exit_code: int | None,
|
||||
how: str,
|
||||
) -> KillOutcome:
|
||||
"""崩溃之后重读一次日志,判尾部形态是不是要的那个时机。
|
||||
|
||||
@@ -759,10 +786,14 @@ def _judge_crash(
|
||||
"""
|
||||
crashed = log_path.read_bytes() if log_path.is_file() else b""
|
||||
after = parse_terminated(crashed)
|
||||
if not should_kill(after, timing=timing, after_steps=after_steps):
|
||||
audited = bool(read_audit_lines(workspace))
|
||||
if not should_kill(after, timing=timing, after_steps=after_steps, audit_is_not_empty=audited):
|
||||
return KillOutcome(
|
||||
hit=False,
|
||||
reason=f"{how}之后重读日志,尾部不是时机 {timing.value} 要的形态",
|
||||
reason=(
|
||||
f"{how}之后重读日志,尾部不是时机 {timing.value} 要的形态"
|
||||
+ ("(审计账还是空的)" if not audited else "")
|
||||
),
|
||||
model_calls=count_model_calls(after),
|
||||
exit_code=exit_code,
|
||||
)
|
||||
@@ -780,10 +811,12 @@ async def spawn_and_kill(
|
||||
*,
|
||||
argv: Sequence[str],
|
||||
log_path: Path,
|
||||
workspace: Path,
|
||||
timing: KillTiming,
|
||||
after_steps: int,
|
||||
child_log_path: Path | None = None,
|
||||
poll_interval_s: float = 0.002,
|
||||
self_kill_grace_s: float = SELF_KILL_GRACE_S,
|
||||
timeout_s: float = 600.0,
|
||||
) -> KillOutcome:
|
||||
"""起一个子进程,等它崩在时机上。
|
||||
@@ -791,10 +824,15 @@ async def spawn_and_kill(
|
||||
**主路径是子进程自己在时机上 `os._exit`**(见 `SelfKillingStore`),父进程只负责认领:
|
||||
退出码等于 `CRASH_EXIT_CODE` 且日志尾部形态对得上,就算命中。
|
||||
|
||||
**条件对上之后先等一个自杀窗口,窗口内子进程还活着才兜底发信号。** 父子两侧的条件现在
|
||||
是同一条,所以父进程看见条件成立的那一刻,子进程正走在自杀那一行上——不留窗口的话父进程
|
||||
每次都抢先,自杀路径成了永远走不到的死代码。实测就是这样:两次崩溃全是外部信号命中的。
|
||||
窗口只用来吸收进程退出的那点延迟,所以取值很小。
|
||||
|
||||
**外部 SIGKILL 那条路留着兜底**,没有删掉:子进程那侧的自杀条件万一因为别的原因没触发
|
||||
(包装漏了一处、审计账一直是空的),轮询仍然会在尾部形态对上的那一刻把它杀掉。**用
|
||||
SIGKILL 不用 SIGTERM**,要的是没有任何清理机会的死法——SIGTERM 会走 Python 的信号处理,
|
||||
`finally` 有机会跑完,那验的是优雅退出而不是崩溃。
|
||||
(包装漏了一处、子进程卡在别的地方),窗口过后仍然会把它杀掉。**用 SIGKILL 不用
|
||||
SIGTERM**,要的是没有任何清理机会的死法——SIGTERM 会走 Python 的信号处理,`finally`
|
||||
有机会跑完,那验的是优雅退出而不是崩溃。
|
||||
|
||||
子进程的输出**写进文件而不是管道**:管道缓冲区满了子进程会阻塞在写上,而表现是「它卡住
|
||||
不动了」,从外面区分不出是卡在模型调用上还是卡在一行日志上。
|
||||
@@ -814,14 +852,11 @@ async def spawn_and_kill(
|
||||
if handle is not None:
|
||||
handle.close()
|
||||
|
||||
deadline = time.monotonic() + timeout_s
|
||||
try:
|
||||
while True:
|
||||
if process.returncode is not None:
|
||||
code = process.returncode
|
||||
def _exited(code: int) -> KillOutcome:
|
||||
if code == CRASH_EXIT_CODE:
|
||||
return _judge_crash(
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=timing,
|
||||
after_steps=after_steps,
|
||||
exit_code=code,
|
||||
@@ -836,15 +871,33 @@ async def spawn_and_kill(
|
||||
model_calls=count_model_calls(read_log(log_path)),
|
||||
exit_code=code,
|
||||
)
|
||||
if should_kill(read_log(log_path), timing=timing, after_steps=after_steps):
|
||||
|
||||
deadline = time.monotonic() + timeout_s
|
||||
try:
|
||||
while True:
|
||||
if process.returncode is not None:
|
||||
return _exited(process.returncode)
|
||||
matched = should_kill(
|
||||
read_log(log_path),
|
||||
timing=timing,
|
||||
after_steps=after_steps,
|
||||
audit_is_not_empty=bool(read_audit_lines(workspace)),
|
||||
)
|
||||
if matched:
|
||||
grace_deadline = time.monotonic() + self_kill_grace_s
|
||||
while process.returncode is None and time.monotonic() < grace_deadline:
|
||||
await asyncio.sleep(poll_interval_s)
|
||||
if process.returncode is not None:
|
||||
return _exited(process.returncode)
|
||||
process.kill()
|
||||
await process.wait()
|
||||
return _judge_crash(
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=timing,
|
||||
after_steps=after_steps,
|
||||
exit_code=process.returncode,
|
||||
how="父进程兜底发了 SIGKILL",
|
||||
how=f"等了 {self_kill_grace_s} 秒不见子进程自杀,父进程兜底发了 SIGKILL",
|
||||
)
|
||||
if time.monotonic() > deadline:
|
||||
return KillOutcome(
|
||||
@@ -1187,6 +1240,7 @@ async def run_crash_fault(
|
||||
model_client: object,
|
||||
guard: CallGuard,
|
||||
attempts: int,
|
||||
self_kill_grace_s: float = SELF_KILL_GRACE_S,
|
||||
) -> FaultReport:
|
||||
"""一类崩溃续跑:起子进程 → 它按时机自杀 → 拷字节 → 同一个 run_id 续跑 → 逐条判。"""
|
||||
task = load_govdoc_task(govdoc_db=govdoc_db, govdoc_corpus=govdoc_corpus)
|
||||
@@ -1224,9 +1278,11 @@ async def run_crash_fault(
|
||||
outcome = await spawn_and_kill(
|
||||
argv=argv,
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=timing,
|
||||
after_steps=CRASH_AFTER_STEPS,
|
||||
child_log_path=runs_dir / f"{run_id}.child.log",
|
||||
self_kill_grace_s=self_kill_grace_s,
|
||||
)
|
||||
guard.charge(outcome.model_calls)
|
||||
if not outcome.hit:
|
||||
@@ -1711,6 +1767,12 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
parser.add_argument(
|
||||
"--attempts", type=int, default=3, help="崩溃续跑每类最多重试几次去命中时机,默认 3"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--self-kill-grace",
|
||||
type=float,
|
||||
default=SELF_KILL_GRACE_S,
|
||||
help=f"条件对上之后留给子进程自杀的秒数,超了才兜底发 SIGKILL,默认 {SELF_KILL_GRACE_S}",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--child", action="store_true", help="内部用:以子进程身份跑一次 GovDoc 运行"
|
||||
)
|
||||
@@ -1790,6 +1852,7 @@ async def run_all(args: argparse.Namespace) -> int:
|
||||
model_client=model_client,
|
||||
guard=guard,
|
||||
attempts=args.attempts,
|
||||
self_kill_grace_s=args.self_kill_grace,
|
||||
),
|
||||
)
|
||||
)
|
||||
@@ -1923,6 +1986,7 @@ __all__ = [
|
||||
"CRASH_EXIT_CODE",
|
||||
"FAULT_NAMES",
|
||||
"GOVDOC_FAULTS",
|
||||
"SELF_KILL_GRACE_S",
|
||||
"AlwaysInvalidParser",
|
||||
"CallGuard",
|
||||
"Criterion",
|
||||
|
||||
+146
-45
@@ -588,32 +588,69 @@ def test_lease_returned_passes_and_breaches() -> None:
|
||||
|
||||
def test_should_kill_after_step_waits_for_enough_steps() -> None:
|
||||
read = as_read(intent(), model_result(), step_completed())
|
||||
assert should_kill(read, timing=KillTiming.AFTER_STEP, after_steps=2) is False
|
||||
assert should_kill(read, timing=KillTiming.AFTER_STEP, after_steps=1) is True
|
||||
assert (
|
||||
should_kill(read, timing=KillTiming.AFTER_STEP, after_steps=2, audit_is_not_empty=True)
|
||||
is False
|
||||
)
|
||||
assert (
|
||||
should_kill(read, timing=KillTiming.AFTER_STEP, after_steps=1, audit_is_not_empty=True)
|
||||
is True
|
||||
)
|
||||
|
||||
|
||||
def test_should_kill_after_step_rejects_trailing_intent() -> None:
|
||||
read = as_read(step_completed(), intent(call_index=1, result_id="r1"))
|
||||
assert should_kill(read, timing=KillTiming.AFTER_STEP, after_steps=1) is False
|
||||
assert (
|
||||
should_kill(read, timing=KillTiming.AFTER_STEP, after_steps=1, audit_is_not_empty=True)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
def test_should_kill_at_intent_requires_never_policy() -> None:
|
||||
"""`safe` 那条意图不算命中:悬在它上面续跑会重放动作接着跑,停止原因不是状态未知。"""
|
||||
never = as_read(step_completed(), intent(call_index=1, result_id="r1", replay_policy="never"))
|
||||
safe = as_read(step_completed(), intent(call_index=1, result_id="r1", replay_policy="safe"))
|
||||
assert should_kill(never, timing=KillTiming.AT_INTENT, after_steps=1) is True
|
||||
assert should_kill(safe, timing=KillTiming.AT_INTENT, after_steps=1) is False
|
||||
assert (
|
||||
should_kill(never, timing=KillTiming.AT_INTENT, after_steps=1, audit_is_not_empty=True)
|
||||
is True
|
||||
)
|
||||
assert (
|
||||
should_kill(safe, timing=KillTiming.AT_INTENT, after_steps=1, audit_is_not_empty=True)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
def test_should_kill_at_intent_rejects_trailing_step() -> None:
|
||||
read = as_read(step_completed())
|
||||
assert should_kill(read, timing=KillTiming.AT_INTENT, after_steps=0) is False
|
||||
assert (
|
||||
should_kill(read, timing=KillTiming.AT_INTENT, after_steps=0, audit_is_not_empty=True)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
def test_should_kill_on_empty_log() -> None:
|
||||
empty = LogRead(payloads=(), torn=False, bad_lines=())
|
||||
assert should_kill(empty, timing=KillTiming.AFTER_STEP, after_steps=0) is False
|
||||
assert should_kill(empty, timing=KillTiming.AT_INTENT, after_steps=0) is False
|
||||
assert (
|
||||
should_kill(empty, timing=KillTiming.AFTER_STEP, after_steps=0, audit_is_not_empty=True)
|
||||
is False
|
||||
)
|
||||
assert (
|
||||
should_kill(empty, timing=KillTiming.AT_INTENT, after_steps=0, audit_is_not_empty=True)
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
def test_should_kill_requires_a_non_empty_audit() -> None:
|
||||
"""父进程的条件必须与子进程的自杀条件逐字对齐,包括审计账那一项。
|
||||
|
||||
松一档的后果实测过:父进程每次都在子进程走到自杀那一行之前抢先发信号,自杀路径成了死代码,
|
||||
而崩溃点落在哪儿又变回碰运气——其中一次就崩在审计账还是空的时候,最硬那条判据只能真空成立。
|
||||
"""
|
||||
after_step = as_read(intent(), model_result(), step_completed())
|
||||
at_intent = as_read(step_completed(), intent(call_index=1, result_id="r1"))
|
||||
for read, timing in ((after_step, KillTiming.AFTER_STEP), (at_intent, KillTiming.AT_INTENT)):
|
||||
assert should_kill(read, timing=timing, after_steps=1, audit_is_not_empty=True) is True
|
||||
assert should_kill(read, timing=timing, after_steps=1, audit_is_not_empty=False) is False
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -707,7 +744,7 @@ def wrapped(
|
||||
"""包一层,并按需要让工作区的审计账非空。"""
|
||||
workspace.mkdir(parents=True, exist_ok=True)
|
||||
if audited:
|
||||
(workspace / AUDIT_LOG_NAME).write_text("write_note\tevidence.md\taaa\n", encoding="utf-8")
|
||||
seed_audit(workspace)
|
||||
return SelfKillingStore(
|
||||
inner=inner, timing=timing, workspace=workspace, exit_now=_exit_sentinel
|
||||
)
|
||||
@@ -828,9 +865,10 @@ with open(path, "a", encoding="utf-8") as handle:
|
||||
handle.write(json.dumps(record) + "\\n")
|
||||
"""
|
||||
|
||||
#: 写完就按崩溃退出码把自己打死,模拟 `SelfKillingStore` 那条主路径。
|
||||
#: 写完就按给定的退出码把自己打死,模拟 `SelfKillingStore` 那条主路径。第四个参数是写完之后
|
||||
#: 先睡多久再死——用来验父进程真的留了自杀窗口,而不是看见条件成立就立刻开枪。
|
||||
_FAKE_CHILD_SELF_KILL = """
|
||||
import json, os, sys
|
||||
import json, os, sys, time
|
||||
path = sys.argv[1]
|
||||
records = json.loads(sys.argv[2])
|
||||
with open(path, "a", encoding="utf-8") as handle:
|
||||
@@ -839,24 +877,39 @@ with open(path, "a", encoding="utf-8") as handle:
|
||||
handle.flush()
|
||||
print("子进程说了句话")
|
||||
sys.stdout.flush()
|
||||
time.sleep(float(sys.argv[4]))
|
||||
os._exit(int(sys.argv[3]))
|
||||
"""
|
||||
|
||||
|
||||
async def test_spawn_and_kill_accepts_a_self_killed_child(tmp_path: Path) -> None:
|
||||
"""主路径:子进程按崩溃退出码自杀,且日志尾部形态对得上,算命中。"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
records = [intent(), model_result(), step_completed()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=[
|
||||
def seed_audit(workspace: Path) -> Path:
|
||||
"""让工作区的审计账非空。父子两侧的命中条件都要求它。"""
|
||||
workspace.mkdir(parents=True, exist_ok=True)
|
||||
(workspace / AUDIT_LOG_NAME).write_text("write_note\tevidence.md\taaa\n", encoding="utf-8")
|
||||
return workspace
|
||||
|
||||
|
||||
def self_kill_argv(log_path: Path, records: list[dict[str, object]], *, code: int, sleep: float):
|
||||
return [
|
||||
sys.executable,
|
||||
"-c",
|
||||
_FAKE_CHILD_SELF_KILL,
|
||||
str(log_path),
|
||||
json.dumps(records),
|
||||
str(CRASH_EXIT_CODE),
|
||||
],
|
||||
str(code),
|
||||
str(sleep),
|
||||
]
|
||||
|
||||
|
||||
async def test_spawn_and_kill_accepts_a_self_killed_child(tmp_path: Path) -> None:
|
||||
"""主路径:子进程按崩溃退出码自杀,且日志尾部形态对得上,算命中。"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = seed_audit(tmp_path / "ws")
|
||||
records = [intent(), model_result(), step_completed()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=self_kill_argv(log_path, records, code=CRASH_EXIT_CODE, sleep=0.0),
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
child_log_path=tmp_path / f"{RUN_ID}.child.log",
|
||||
@@ -870,20 +923,82 @@ async def test_spawn_and_kill_accepts_a_self_killed_child(tmp_path: Path) -> Non
|
||||
assert "子进程说了句话" in (tmp_path / f"{RUN_ID}.child.log").read_text(encoding="utf-8")
|
||||
|
||||
|
||||
async def test_spawn_and_kill_waits_out_the_self_kill_grace(tmp_path: Path) -> None:
|
||||
"""条件对上之后子进程还要过一会儿才死:父进程必须等它,不许抢先开枪。
|
||||
|
||||
抢先的后果不是「杀错了」,是自杀那条路永远走不到,而崩溃点落在哪儿又变回碰运气。
|
||||
"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = seed_audit(tmp_path / "ws")
|
||||
records = [intent(), model_result(), step_completed()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=self_kill_argv(log_path, records, code=CRASH_EXIT_CODE, sleep=0.4),
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
self_kill_grace_s=5.0,
|
||||
timeout_s=20.0,
|
||||
)
|
||||
assert outcome.hit is True
|
||||
assert outcome.exit_code == CRASH_EXIT_CODE
|
||||
assert "自杀" in outcome.reason
|
||||
|
||||
|
||||
async def test_spawn_and_kill_falls_back_after_the_grace_runs_out(tmp_path: Path) -> None:
|
||||
"""窗口用完子进程还活着,才轮到兜底 SIGKILL。这条路径没被删。"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = seed_audit(tmp_path / "ws")
|
||||
records = [intent(), model_result(), step_completed()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=self_kill_argv(log_path, records, code=CRASH_EXIT_CODE, sleep=30.0),
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
self_kill_grace_s=0.2,
|
||||
timeout_s=20.0,
|
||||
)
|
||||
assert outcome.hit is True
|
||||
assert "兜底" in outcome.reason
|
||||
assert outcome.exit_code == -9
|
||||
assert outcome.steps == 1
|
||||
assert outcome.model_calls == 1
|
||||
assert outcome.snapshot == log_path.read_bytes()
|
||||
|
||||
|
||||
async def test_spawn_and_kill_needs_a_non_empty_audit(tmp_path: Path) -> None:
|
||||
"""日志尾部形态对了但审计账是空的:不算命中,等到超时为止。
|
||||
|
||||
这一条守的正是上一版实测出来的毛病——那次崩在第 1 步、审计账 0 条,最硬那条判据只能报
|
||||
无法判定,而报告上看起来只是「少了一条」。
|
||||
"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = tmp_path / "ws"
|
||||
workspace.mkdir()
|
||||
records = [intent(), model_result(), step_completed()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=[sys.executable, "-c", _FAKE_CHILD, str(log_path), json.dumps(records)],
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
self_kill_grace_s=0.2,
|
||||
timeout_s=0.6,
|
||||
)
|
||||
assert outcome.hit is False
|
||||
assert "仍没崩在时机" in outcome.reason
|
||||
|
||||
|
||||
async def test_spawn_and_kill_rejects_a_self_kill_at_the_wrong_tail(tmp_path: Path) -> None:
|
||||
"""自杀了但尾部形态不对:报没命中,不许因为退出码对就当成命中。"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = seed_audit(tmp_path / "ws")
|
||||
records = [intent(), model_result()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=[
|
||||
sys.executable,
|
||||
"-c",
|
||||
_FAKE_CHILD_SELF_KILL,
|
||||
str(log_path),
|
||||
json.dumps(records),
|
||||
str(CRASH_EXIT_CODE),
|
||||
],
|
||||
argv=self_kill_argv(log_path, records, code=CRASH_EXIT_CODE, sleep=0.0),
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
timeout_s=20.0,
|
||||
@@ -892,31 +1007,15 @@ async def test_spawn_and_kill_rejects_a_self_kill_at_the_wrong_tail(tmp_path: Pa
|
||||
assert "尾部不是时机" in outcome.reason
|
||||
|
||||
|
||||
async def test_spawn_and_kill_still_falls_back_to_sigkill(tmp_path: Path) -> None:
|
||||
"""兜底路径没删:子进程一直不自杀时,父进程仍然会在尾部形态对上的那一刻杀掉它。"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
records = [intent(), model_result(), step_completed()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=[sys.executable, "-c", _FAKE_CHILD, str(log_path), json.dumps(records)],
|
||||
log_path=log_path,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
timeout_s=20.0,
|
||||
)
|
||||
assert outcome.hit is True
|
||||
assert "兜底" in outcome.reason
|
||||
assert outcome.steps == 1
|
||||
assert outcome.model_calls == 1
|
||||
assert outcome.snapshot == log_path.read_bytes()
|
||||
|
||||
|
||||
async def test_spawn_and_kill_reports_a_miss_when_the_child_exits_normally(tmp_path: Path) -> None:
|
||||
"""子进程正常跑完了要报出来,不许悄悄当成命中。"""
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = seed_audit(tmp_path / "ws")
|
||||
records = [intent(), model_result()]
|
||||
outcome = await spawn_and_kill(
|
||||
argv=[sys.executable, "-c", _FAKE_CHILD_EXITS, str(log_path), json.dumps(records)],
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
timeout_s=20.0,
|
||||
@@ -928,9 +1027,11 @@ async def test_spawn_and_kill_reports_a_miss_when_the_child_exits_normally(tmp_p
|
||||
|
||||
async def test_spawn_and_kill_reports_a_miss_on_timeout(tmp_path: Path) -> None:
|
||||
log_path = tmp_path / f"{RUN_ID}.jsonl"
|
||||
workspace = seed_audit(tmp_path / "ws")
|
||||
outcome = await spawn_and_kill(
|
||||
argv=[sys.executable, "-c", _FAKE_CHILD, str(log_path), json.dumps([intent()])],
|
||||
log_path=log_path,
|
||||
workspace=workspace,
|
||||
timing=KillTiming.AFTER_STEP,
|
||||
after_steps=1,
|
||||
timeout_s=0.5,
|
||||
|
||||
Reference in New Issue
Block a user