feat(soak): GovDoc 形态的场景——JSON 工具调用、按次收窄、提交型完成
这条路径至今零真实负载,而它和 AppWorld 那路压的东西完全不同:动作是 JSON 工具调用而 不是代码,完成由 agent 自报(带完成标记的提交工具)而不是环境报告,工具集按阶段收窄。 一个任务拆成三次运行(plan / execute / summarize),因为「按次收窄工具集」在这个库里 只能这么表达——治理单位是一次运行,一次运行只有一个工具注册表,阶段之间靠工作区传状态。 语料是真实公文,但先脱敏再用,且原文与脱敏文本都只在内存里,一个字节不落盘、不入库。 机构名、电话、信用代码、邮箱、联系人姓名换成明显是假的稳定假名(同一原名整份文档换成 同一假名,否则模型会以为是不同主体);金额、项目编号、日期原样保留,它们是审核判断的 依据,换掉任务就没得判了。脱敏后必须过一遍独立的检出校验,有残留就拒绝启动——把未脱敏 的第三方真实信息发给外部模型服务是不可逆的,而漏一处的表现是「压测正常跑完」。 实测:主招标文件 17.5 万字符,校验函数对原文报 61 处、对脱敏后放行,六份语料全过, 40 个金额与全部日期无误伤。 write_note 与 submit_finding 声明为绝不重放,每次调用在工作区留一行审计——那份审计是 环境侧的实际执行次数证据,故障注入要验的「绝不重放的动作没有被执行两次」数的就是它。 Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
This commit is contained in:
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,568 @@
|
|||||||
|
"""GovDoc 公文审核场景适配器的测试。
|
||||||
|
|
||||||
|
分五块:脱敏、解析器、工具、阶段收窄、装配。
|
||||||
|
|
||||||
|
脱敏那块里最重要的一条是「校验函数对未脱敏文本确实会抛异常」。一个永远返回通过的校验函数比
|
||||||
|
没有校验更糟:它会让所有人以为这道闸在守着,而它什么都没守。
|
||||||
|
|
||||||
|
解析器那块的前五条是 `tests/contract/test_decision_parser.py` 那份公共契约的逐条复刻。契约
|
||||||
|
套件本身是给下游接自己的实现用的(在自己的 `conftest.py` 里覆盖 fixture),压测这边不接那套
|
||||||
|
装配、只把五条断言照着写一遍——它是任何新适配器的准入标准,压测的适配器也是适配器。
|
||||||
|
|
||||||
|
用真实数据的那几条在数据目录不在时跳过而不是失败:那份数据是另一个项目的工作副本,不在本仓库
|
||||||
|
里,换一台机器就没有。
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
from typing import TYPE_CHECKING
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from polyloop.ports import Action, InvalidDecision
|
||||||
|
from polyloop.tools import ToolRegistry
|
||||||
|
from polyloop.types import ModelReply, ReplayPolicy
|
||||||
|
from tools.soak.scenarios.govdoc import (
|
||||||
|
AUDIT_LOG_NAME,
|
||||||
|
DEFAULT_DATA_ROOT,
|
||||||
|
FINDING_NAME,
|
||||||
|
MAX_READ_LINES,
|
||||||
|
PHASE_TOOLS,
|
||||||
|
VERDICT_LEVELS,
|
||||||
|
AuditTask,
|
||||||
|
Checkpoint,
|
||||||
|
CorpusDocument,
|
||||||
|
GovDocParser,
|
||||||
|
GovDocScenarioError,
|
||||||
|
GovDocTools,
|
||||||
|
RedactionResidueError,
|
||||||
|
Redactor,
|
||||||
|
assert_no_residue,
|
||||||
|
build_audit_tasks,
|
||||||
|
build_context,
|
||||||
|
build_run_request,
|
||||||
|
make_run_id,
|
||||||
|
read_audit_lines,
|
||||||
|
)
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
#: 一段自造的「像真的」文本:机构全称、医院、财政局、固定电话、统一社会信用代码、邮箱、
|
||||||
|
#: 联系人各一处,同一家公司出现两次。
|
||||||
|
#:
|
||||||
|
#: **全部是编造的,一个字都不取自真实文书**:机构名前面带「虚构」两字,邮箱域名带 example,
|
||||||
|
#: 号码是连号。测试数据本身要是可识别的,那这份测试就成了它自己要挡的那种泄漏。
|
||||||
|
DIRTY_TEXT = """项目名称:某设备采购
|
||||||
|
采购人:虚构市第三人民医院
|
||||||
|
采购代理机构:虚构鸿远工程咨询有限公司
|
||||||
|
监督部门:虚构市财政局
|
||||||
|
代理机构地址:虚构市朝阳街道 88 号
|
||||||
|
联系人:赵明
|
||||||
|
电话:0768-12345678
|
||||||
|
邮箱:zhaoming@example-invalid.cn
|
||||||
|
统一社会信用代码:91445102MA4XK7YQ3B
|
||||||
|
中标供应商:虚构鸿远工程咨询有限公司
|
||||||
|
预算金额:8,736,100.00 元
|
||||||
|
项目编号:440513-2023-03374
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def _reply(content: str) -> ModelReply:
|
||||||
|
return ModelReply(call_id="call-1", content=content, thinking="")
|
||||||
|
|
||||||
|
|
||||||
|
def _sample_task() -> AuditTask:
|
||||||
|
checkpoint = Checkpoint(
|
||||||
|
checkpoint_id="cp-1",
|
||||||
|
category="不合理条件限制或排斥供应商",
|
||||||
|
title="1.直接或变相对外地企业进入本地市场设置阻碍。",
|
||||||
|
description="采购文件设置供应商注册地等不合理的资格条件、评审因素。",
|
||||||
|
legal_basis=("政府采购法第5条", "第22条第二款"),
|
||||||
|
severity="major",
|
||||||
|
)
|
||||||
|
document = CorpusDocument.from_text(
|
||||||
|
logical_name="tender.md",
|
||||||
|
text="\n".join(f"第 {number} 行:投标人须在本地注册。" for number in range(1, 1001)),
|
||||||
|
)
|
||||||
|
return AuditTask(index=0, checkpoint=checkpoint, documents=(document,))
|
||||||
|
|
||||||
|
|
||||||
|
def _tools(tmp_path: Path) -> GovDocTools:
|
||||||
|
return GovDocTools(documents=_sample_task().documents, workspace=tmp_path)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# 一、脱敏
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_validator_rejects_unredacted_text():
|
||||||
|
"""这条是这份测试里最要紧的一条:校验函数必须真的会拒。"""
|
||||||
|
with pytest.raises(RedactionResidueError):
|
||||||
|
assert_no_residue(DIRTY_TEXT, where="自造样本")
|
||||||
|
|
||||||
|
|
||||||
|
def test_validator_accepts_redacted_text():
|
||||||
|
result = Redactor().redact(DIRTY_TEXT)
|
||||||
|
assert_no_residue(result.text, where="自造样本")
|
||||||
|
|
||||||
|
|
||||||
|
def test_every_identifier_category_is_replaced():
|
||||||
|
result = Redactor().redact(DIRTY_TEXT)
|
||||||
|
for category in ("company", "hospital", "bureau", "phone", "email", "uscc", "person"):
|
||||||
|
assert result.counts.get(category, 0) >= 1, f"{category} 一处都没替换:{result.counts}"
|
||||||
|
for leaked in (
|
||||||
|
"虚构鸿远工程咨询有限公司",
|
||||||
|
"虚构市第三人民医院",
|
||||||
|
"虚构市财政局",
|
||||||
|
"0768-12345678",
|
||||||
|
"zhaoming@example-invalid.cn",
|
||||||
|
"91445102MA4XK7YQ3B",
|
||||||
|
"赵明",
|
||||||
|
):
|
||||||
|
assert leaked not in result.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_same_original_gets_one_stable_alias():
|
||||||
|
result = Redactor().redact(DIRTY_TEXT)
|
||||||
|
# 那家公司在原文里出现两次(第 3 行的代理机构、倒数第 3 行的中标供应商),
|
||||||
|
# 替换后必须还是同一个假名、同样两次。
|
||||||
|
company_alias = result.text.splitlines()[2].split(":", 1)[1]
|
||||||
|
assert company_alias.startswith("示例")
|
||||||
|
assert result.text.count(company_alias) == 2
|
||||||
|
assert result.distinct["company"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_alias_is_stable_across_documents():
|
||||||
|
redactor = Redactor()
|
||||||
|
first = redactor.redact(DIRTY_TEXT).text
|
||||||
|
second = redactor.redact("中标人是虚构鸿远工程咨询有限公司。").text
|
||||||
|
alias = first.splitlines()[2].split(":", 1)[1]
|
||||||
|
assert alias in second
|
||||||
|
|
||||||
|
|
||||||
|
def test_amounts_and_project_numbers_survive():
|
||||||
|
result = Redactor().redact(DIRTY_TEXT)
|
||||||
|
assert "8,736,100.00" in result.text
|
||||||
|
assert "440513-2023-03374" in result.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_generic_institution_words_survive():
|
||||||
|
text = "评标委员会依法组建,投标人可由总公司授权分公司投标。"
|
||||||
|
result = Redactor().redact(text)
|
||||||
|
assert result.text == text
|
||||||
|
|
||||||
|
|
||||||
|
def test_redaction_is_idempotent():
|
||||||
|
once = Redactor().redact(DIRTY_TEXT).text
|
||||||
|
twice = Redactor().redact(once).text
|
||||||
|
assert twice == once
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# 二、决策解释器:五条公共契约 + 两种容错 + 五种失败
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
_ACTION_REPLY = (
|
||||||
|
'{"tool": "read_document", "arguments": {"path": "tender.md", "start_line": 1, "end_line": 20}}'
|
||||||
|
)
|
||||||
|
_INVALID_REPLY = "我先想想应该从哪里开始查。"
|
||||||
|
|
||||||
|
|
||||||
|
def test_contract_1_parse_is_synchronous():
|
||||||
|
parsed = GovDocParser().parse(_reply(_ACTION_REPLY))
|
||||||
|
assert not hasattr(parsed, "__await__")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("content", [_ACTION_REPLY, _INVALID_REPLY, "", "```json\n{}\n```"])
|
||||||
|
def test_contract_2_history_text_is_not_longer(content: str):
|
||||||
|
parsed = GovDocParser().parse(_reply(content))
|
||||||
|
assert isinstance(parsed.history_text, str)
|
||||||
|
assert len(parsed.history_text) <= len(content)
|
||||||
|
|
||||||
|
|
||||||
|
def test_contract_3_invalid_decision_explains_itself():
|
||||||
|
parsed = GovDocParser().parse(_reply(_INVALID_REPLY))
|
||||||
|
assert isinstance(parsed.decision, InvalidDecision)
|
||||||
|
assert parsed.decision.explanation.strip()
|
||||||
|
|
||||||
|
|
||||||
|
def test_contract_4_action_carries_text():
|
||||||
|
parsed = GovDocParser().parse(_reply(_ACTION_REPLY))
|
||||||
|
assert isinstance(parsed.decision, Action)
|
||||||
|
assert isinstance(parsed.decision.text, str)
|
||||||
|
assert parsed.decision.text
|
||||||
|
assert parsed.decision.tool_call is not None
|
||||||
|
assert parsed.decision.tool_call.name == "read_document"
|
||||||
|
assert parsed.decision.tool_call.arguments["path"] == "tender.md"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"content",
|
||||||
|
[
|
||||||
|
"",
|
||||||
|
" ",
|
||||||
|
"{",
|
||||||
|
"```json\n",
|
||||||
|
"[1, 2, 3]",
|
||||||
|
'{"tool": null}',
|
||||||
|
'{"tool": "x", "arguments": 5}',
|
||||||
|
"```\n```",
|
||||||
|
"\x00\x01",
|
||||||
|
"{" * 500,
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_contract_5_parse_never_raises(content: str):
|
||||||
|
parsed = GovDocParser().parse(_reply(content))
|
||||||
|
assert isinstance(parsed.decision, Action | InvalidDecision)
|
||||||
|
|
||||||
|
|
||||||
|
def test_tolerates_code_fences():
|
||||||
|
for fenced in (
|
||||||
|
f"```json\n{_ACTION_REPLY}\n```",
|
||||||
|
f"```\n{_ACTION_REPLY}\n```",
|
||||||
|
f"我打算先读一段:\n```json\n{_ACTION_REPLY}\n```\n读完再说。",
|
||||||
|
):
|
||||||
|
parsed = GovDocParser().parse(_reply(fenced))
|
||||||
|
assert isinstance(parsed.decision, Action), fenced
|
||||||
|
assert parsed.decision.tool_call is not None
|
||||||
|
assert parsed.decision.tool_call.name == "read_document"
|
||||||
|
|
||||||
|
|
||||||
|
def test_tolerates_flattened_arguments():
|
||||||
|
parsed = GovDocParser().parse(
|
||||||
|
_reply('{"tool": "read_document", "path": "tender.md", "start_line": 1, "end_line": 20}')
|
||||||
|
)
|
||||||
|
assert isinstance(parsed.decision, Action)
|
||||||
|
assert parsed.decision.tool_call is not None
|
||||||
|
assert parsed.decision.tool_call.arguments == {
|
||||||
|
"path": "tender.md",
|
||||||
|
"start_line": 1,
|
||||||
|
"end_line": 20,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_five_parse_failures_get_five_explanations():
|
||||||
|
parser = GovDocParser()
|
||||||
|
explanations = {}
|
||||||
|
for label, content in {
|
||||||
|
"没有 JSON": "我准备开始审核了。",
|
||||||
|
"语法错": '{"tool": "read_document", "arguments": {,}}',
|
||||||
|
"不是对象": "[1, 2, 3]",
|
||||||
|
"缺 tool": '{"arguments": {"path": "tender.md"}}',
|
||||||
|
"arguments 不是对象": '{"tool": "read_document", "arguments": "tender.md"}',
|
||||||
|
}.items():
|
||||||
|
decision = parser.parse(_reply(content)).decision
|
||||||
|
assert isinstance(decision, InvalidDecision), label
|
||||||
|
explanations[label] = decision.explanation
|
||||||
|
assert len(set(explanations.values())) == 5, explanations
|
||||||
|
assert "语法" in explanations["语法错"]
|
||||||
|
assert "tool" in explanations["缺 tool"]
|
||||||
|
assert "arguments" in explanations["arguments 不是对象"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_parser_reports_its_parameters():
|
||||||
|
assert GovDocParser().parameters() == {"kind": "govdoc_json_tool_call"}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# 三、四个工具
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
async def test_read_document_returns_numbered_lines(tmp_path: Path):
|
||||||
|
body = await _tools(tmp_path).read_document(
|
||||||
|
{"path": "tender.md", "start_line": 3, "end_line": 5}
|
||||||
|
)
|
||||||
|
assert body.splitlines() == [
|
||||||
|
"3: 第 3 行:投标人须在本地注册。",
|
||||||
|
"4: 第 4 行:投标人须在本地注册。",
|
||||||
|
"5: 第 5 行:投标人须在本地注册。",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_read_document_truncates_and_says_so(tmp_path: Path):
|
||||||
|
body = await _tools(tmp_path).read_document(
|
||||||
|
{"path": "tender.md", "start_line": 1, "end_line": 1000}
|
||||||
|
)
|
||||||
|
lines = body.splitlines()
|
||||||
|
assert len(lines) == MAX_READ_LINES + 1
|
||||||
|
assert lines[MAX_READ_LINES - 1].startswith(f"{MAX_READ_LINES}: ")
|
||||||
|
assert f"第 1 到第 {MAX_READ_LINES} 行" in lines[-1]
|
||||||
|
assert f"还剩 {1000 - MAX_READ_LINES} 行未返回" in lines[-1]
|
||||||
|
assert "共 1000 行" in lines[-1]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"path", ["../etc/passwd", "/etc/passwd", "notes/plan.md", "nope.md", ".write_audit.log"]
|
||||||
|
)
|
||||||
|
async def test_read_document_rejects_unregistered_path(tmp_path: Path, path: str):
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await _tools(tmp_path).read_document({"path": path, "start_line": 1, "end_line": 2})
|
||||||
|
|
||||||
|
|
||||||
|
async def test_read_document_reads_workspace_notes(tmp_path: Path):
|
||||||
|
tools = _tools(tmp_path)
|
||||||
|
await tools.write_note({"filename": "plan.md", "content": "第一条:查注册地要求"})
|
||||||
|
body = await tools.read_document({"path": "plan.md", "start_line": 1, "end_line": 10})
|
||||||
|
assert "第一条:查注册地要求" in body
|
||||||
|
|
||||||
|
|
||||||
|
async def test_grep_document_caps_matches(tmp_path: Path):
|
||||||
|
tools = _tools(tmp_path)
|
||||||
|
body = await tools.grep_document({"pattern": "本地注册", "path": "tender.md", "max_matches": 3})
|
||||||
|
lines = body.splitlines()
|
||||||
|
assert lines[0].startswith("1: ")
|
||||||
|
assert len(lines) == 4
|
||||||
|
assert "共 1000 处命中" in lines[-1]
|
||||||
|
|
||||||
|
|
||||||
|
async def test_grep_document_raises_on_bad_pattern(tmp_path: Path):
|
||||||
|
# 报错文本要指向正则本身。断言这一句是为了挡住「因为别的参数报错而恰好也抛了 ValueError」
|
||||||
|
# 那种假绿——这条曾经真的因为可选参数 max_matches 没传而在别处先炸掉。
|
||||||
|
with pytest.raises(ValueError, match="正则"):
|
||||||
|
await _tools(tmp_path).grep_document({"pattern": "([", "path": "tender.md"})
|
||||||
|
|
||||||
|
|
||||||
|
async def test_grep_document_max_matches_is_optional(tmp_path: Path):
|
||||||
|
body = await _tools(tmp_path).grep_document({"pattern": "本地注册", "path": "tender.md"})
|
||||||
|
assert len(body.splitlines()) == 51
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("filename", ["../escape.md", "sub/plan.md", "", ".hidden"])
|
||||||
|
async def test_write_note_rejects_paths(tmp_path: Path, filename: str):
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await _tools(tmp_path).write_note({"filename": filename, "content": "x"})
|
||||||
|
|
||||||
|
|
||||||
|
async def test_write_note_writes_and_audits(tmp_path: Path):
|
||||||
|
tools = _tools(tmp_path)
|
||||||
|
await tools.write_note({"filename": "plan.md", "content": "第一版"})
|
||||||
|
await tools.write_note({"filename": "plan.md", "content": "第二版"})
|
||||||
|
assert (tmp_path / "plan.md").read_text(encoding="utf-8") == "第二版"
|
||||||
|
audit = read_audit_lines(tmp_path)
|
||||||
|
assert len(audit) == 2
|
||||||
|
assert all(line.startswith("write_note\tplan.md\t") for line in audit)
|
||||||
|
# 两次内容不同,摘要也要不同——审计要能区分「同一个动作被执行了两次」与「两次写的是同一份」。
|
||||||
|
assert audit[0] != audit[1]
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("verdict", ["合格", "compliant", "", "合规 "])
|
||||||
|
async def test_submit_finding_rejects_bad_verdict(tmp_path: Path, verdict: str):
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await _tools(tmp_path).submit_finding(
|
||||||
|
{"verdict": verdict, "evidence": "第 3 行", "reasoning": "无"}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_submit_finding_writes_and_audits(tmp_path: Path):
|
||||||
|
tools = _tools(tmp_path)
|
||||||
|
for verdict in VERDICT_LEVELS:
|
||||||
|
await tools.submit_finding(
|
||||||
|
{"verdict": verdict, "evidence": "tender.md 第 3 行", "reasoning": "见证据"}
|
||||||
|
)
|
||||||
|
payload = json.loads((tmp_path / FINDING_NAME).read_text(encoding="utf-8"))
|
||||||
|
assert payload["verdict"] == VERDICT_LEVELS[-1]
|
||||||
|
audit = read_audit_lines(tmp_path)
|
||||||
|
assert len(audit) == len(VERDICT_LEVELS)
|
||||||
|
assert all(line.startswith(f"submit_finding\t{FINDING_NAME}\t") for line in audit)
|
||||||
|
|
||||||
|
|
||||||
|
async def test_audit_log_is_not_readable_by_the_model(tmp_path: Path):
|
||||||
|
"""审计是环境侧的账,不是给模型看的材料。"""
|
||||||
|
tools = _tools(tmp_path)
|
||||||
|
await tools.write_note({"filename": "plan.md", "content": "x"})
|
||||||
|
assert (tmp_path / AUDIT_LOG_NAME).is_file()
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await tools.read_document({"path": AUDIT_LOG_NAME, "start_line": 1, "end_line": 2})
|
||||||
|
|
||||||
|
|
||||||
|
async def test_wrong_argument_types_raise_plain_errors(tmp_path: Path):
|
||||||
|
tools = _tools(tmp_path)
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await tools.read_document({"path": "tender.md", "start_line": True, "end_line": 2})
|
||||||
|
with pytest.raises(ValueError):
|
||||||
|
await tools.read_document({"path": 3, "start_line": 1, "end_line": 2})
|
||||||
|
|
||||||
|
|
||||||
|
def test_replay_policies_and_completion_flag(tmp_path: Path):
|
||||||
|
registry = _tools(tmp_path).registry()
|
||||||
|
assert registry.names() == (
|
||||||
|
"read_document",
|
||||||
|
"grep_document",
|
||||||
|
"write_note",
|
||||||
|
"submit_finding",
|
||||||
|
)
|
||||||
|
policies = {name: registry.spec_for(name).replay_policy for name in registry.names()}
|
||||||
|
assert policies == {
|
||||||
|
"read_document": ReplayPolicy.SAFE,
|
||||||
|
"grep_document": ReplayPolicy.SAFE,
|
||||||
|
"write_note": ReplayPolicy.NEVER,
|
||||||
|
"submit_finding": ReplayPolicy.NEVER,
|
||||||
|
}
|
||||||
|
completes = [name for name in registry.names() if registry.spec_for(name).completes_run]
|
||||||
|
assert completes == ["submit_finding"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_tool_parameters_are_closed_json_schema(tmp_path: Path):
|
||||||
|
for spec in _tools(tmp_path).registry().schema_for_model():
|
||||||
|
parameters = spec["parameters"]
|
||||||
|
assert parameters["type"] == "object"
|
||||||
|
assert parameters["additionalProperties"] is False
|
||||||
|
assert parameters["required"]
|
||||||
|
json.dumps(spec, ensure_ascii=False)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# 四、阶段收窄
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_summarize_drops_grep_keeps_submit(tmp_path: Path):
|
||||||
|
full = _tools(tmp_path).registry()
|
||||||
|
narrowed = full.restrict_to(PHASE_TOOLS["summarize"])
|
||||||
|
assert "grep_document" not in narrowed.names()
|
||||||
|
assert "write_note" not in narrowed.names()
|
||||||
|
assert "submit_finding" in narrowed.names()
|
||||||
|
assert "read_document" in narrowed.names()
|
||||||
|
|
||||||
|
|
||||||
|
def test_restrict_to_leaves_the_source_registry_alone(tmp_path: Path):
|
||||||
|
full = _tools(tmp_path).registry()
|
||||||
|
before = full.names()
|
||||||
|
full.restrict_to(PHASE_TOOLS["summarize"])
|
||||||
|
assert full.names() == before
|
||||||
|
assert len(before) == 4
|
||||||
|
|
||||||
|
|
||||||
|
def test_plan_and_execute_have_no_submit_tool(tmp_path: Path):
|
||||||
|
full = _tools(tmp_path).registry()
|
||||||
|
for phase in ("plan", "execute"):
|
||||||
|
narrowed = full.restrict_to(PHASE_TOOLS[phase])
|
||||||
|
assert "submit_finding" not in narrowed.names()
|
||||||
|
assert narrowed.names() == ("read_document", "grep_document", "write_note")
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# 五、装配
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_run_id_shape():
|
||||||
|
assert make_run_id(task_index=7, phase="plan") == "govdoc-7-plan"
|
||||||
|
for phase in ("plan", "execute", "summarize"):
|
||||||
|
assert re.fullmatch(r"[A-Za-z0-9._\-]+", make_run_id(task_index=12, phase=phase))
|
||||||
|
with pytest.raises(GovDocScenarioError):
|
||||||
|
make_run_id(task_index=7, phase="finalize")
|
||||||
|
|
||||||
|
|
||||||
|
def test_all_three_phases_assemble(tmp_path: Path):
|
||||||
|
task = _sample_task()
|
||||||
|
for phase in ("plan", "execute", "summarize"):
|
||||||
|
request = build_run_request(
|
||||||
|
task=task,
|
||||||
|
phase=phase,
|
||||||
|
run_id=make_run_id(task_index=task.index, phase=phase),
|
||||||
|
workspace=tmp_path,
|
||||||
|
model_binding={"session_id": "soak-1"},
|
||||||
|
)
|
||||||
|
# 库会校验执行器与本次可见注册表同源(session/__init__.py:156-170)。
|
||||||
|
assert request.action_executor.registry == request.tools
|
||||||
|
assert request.tools.names() == PHASE_TOOLS[phase]
|
||||||
|
assert "{observation}" in request.observation_template
|
||||||
|
assert request.model_replay_policy is ReplayPolicy.NEVER
|
||||||
|
assert request.cancel_grace_seconds == 5.0
|
||||||
|
assert request.injections == {}
|
||||||
|
|
||||||
|
|
||||||
|
def test_budgets_follow_gov_auditor_turn_limits(tmp_path: Path):
|
||||||
|
task = _sample_task()
|
||||||
|
budgets = {
|
||||||
|
phase: build_run_request(
|
||||||
|
task=task,
|
||||||
|
phase=phase,
|
||||||
|
run_id=make_run_id(task_index=0, phase=phase),
|
||||||
|
workspace=tmp_path,
|
||||||
|
model_binding={},
|
||||||
|
).budget
|
||||||
|
for phase in ("plan", "execute", "summarize")
|
||||||
|
}
|
||||||
|
assert (budgets["plan"].max_steps, budgets["plan"].max_actions) == (50, 50)
|
||||||
|
assert (budgets["execute"].max_steps, budgets["execute"].max_actions) == (50, 50)
|
||||||
|
assert (budgets["summarize"].max_steps, budgets["summarize"].max_actions) == (16, 16)
|
||||||
|
for budget in budgets.values():
|
||||||
|
assert budget.max_consecutive_parse_failures == 3
|
||||||
|
assert budget.max_prompt_chars == 400_000
|
||||||
|
|
||||||
|
|
||||||
|
def test_context_carries_tools_but_not_the_document_body(tmp_path: Path):
|
||||||
|
task = _sample_task()
|
||||||
|
registry = _tools(tmp_path).registry().restrict_to(PHASE_TOOLS["summarize"])
|
||||||
|
context = build_context(task=task, phase="summarize", tools=registry)
|
||||||
|
system = context.run_level[0].content[0].text
|
||||||
|
goal = context.goal_level[0].content[0].text
|
||||||
|
# 提示词正文里会点名说「本阶段没有检索工具」,所以工具清单要在 schema 那一段里查。
|
||||||
|
assert '"name": "submit_finding"' in system
|
||||||
|
assert '"name": "grep_document"' not in system
|
||||||
|
assert task.checkpoint.title in goal
|
||||||
|
assert "共 1000 行" in goal
|
||||||
|
# 公文正文必须靠工具读,不能整篇塞进上下文。
|
||||||
|
assert task.documents[0].lines[0] not in system + goal
|
||||||
|
|
||||||
|
|
||||||
|
async def test_phases_share_state_through_the_workspace(tmp_path: Path):
|
||||||
|
task = _sample_task()
|
||||||
|
plan_tools = GovDocTools(documents=task.documents, workspace=tmp_path)
|
||||||
|
await plan_tools.write_note({"filename": "evidence.md", "content": "证据一:第 3 行"})
|
||||||
|
summarize_tools = GovDocTools(documents=task.documents, workspace=tmp_path)
|
||||||
|
body = await summarize_tools.read_document(
|
||||||
|
{"path": "evidence.md", "start_line": 1, "end_line": 5}
|
||||||
|
)
|
||||||
|
assert "证据一" in body
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_registry_restricts_to_empty(tmp_path: Path):
|
||||||
|
assert ToolRegistry(()).restrict_to(()).names() == ()
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# 六、真实数据(数据目录不在就跳过)
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
_HAS_REAL_DATA = DEFAULT_DATA_ROOT.is_dir()
|
||||||
|
_needs_real_data = pytest.mark.skipif(
|
||||||
|
not _HAS_REAL_DATA, reason=f"数据源不在这台机器上:{DEFAULT_DATA_ROOT}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@_needs_real_data
|
||||||
|
def test_real_corpus_assembles_and_passes_the_gate():
|
||||||
|
tasks, reports = build_audit_tasks(checkpoint_count=3)
|
||||||
|
assert len(tasks) == 3
|
||||||
|
assert reports
|
||||||
|
document = tasks[0].documents[0]
|
||||||
|
assert document.logical_name == "tender.md"
|
||||||
|
assert document.line_count > 2000
|
||||||
|
assert document.char_count > 100_000
|
||||||
|
# 装配路径上已经调过校验函数,这里再自己确认一遍:这道闸是压测能不能启动的判据。
|
||||||
|
assert_no_residue("\n".join(document.lines), where="真实语料")
|
||||||
|
assert_no_residue(tasks[0].checkpoint.render(), where="真实审核点")
|
||||||
|
replaced = sum(count for report in reports for count in report.counts.values())
|
||||||
|
assert replaced > 0
|
||||||
|
|
||||||
|
|
||||||
|
@_needs_real_data
|
||||||
|
def test_real_task_assembles_three_requests(tmp_path: Path):
|
||||||
|
tasks, _ = build_audit_tasks(checkpoint_count=1)
|
||||||
|
for phase in ("plan", "execute", "summarize"):
|
||||||
|
request = build_run_request(
|
||||||
|
task=tasks[0],
|
||||||
|
phase=phase,
|
||||||
|
run_id=make_run_id(task_index=0, phase=phase),
|
||||||
|
workspace=tmp_path,
|
||||||
|
model_binding={},
|
||||||
|
)
|
||||||
|
assert request.action_executor.registry == request.tools
|
||||||
Reference in New Issue
Block a user