test: stop reading channel outages as library defects in live e2e
L9's "unknown shape" sample was the openai profile, which 1.3.3 gave a real shape (off/on_base/effort_key all set), so the guard had nothing to reject. It now registers a shapeless provider of its own and tests the mechanism rather than whichever profile happens to be blank that month. L8 checks the reported model before judging the capability table: this channel answers glm-5 / glm-5.1 / glm-5.2 with glm-5.3, which is a routing problem the library already warns about, not drift. All three are guarded, including the one that passed by luck. T10 tells 404 model_not_found (the channel dropped the model) apart from 400 (the tier really is refused), reading the status code and the body's type field rather than the whole message; only the latter still counts as a conclusion about a tier. An all-skipped tier list now skips instead of going green. TestMiniMaxM3 gained the unreachable fallback its own docstring promised: an outage now skips and leaves an uncovered row, where before it failed ahead of _record and left no trace of what happened.
This commit is contained in:
+168
-29
@@ -47,6 +47,7 @@ from polygateway.errors import (
|
||||
SourceDeadError,
|
||||
TransientError,
|
||||
)
|
||||
from polygateway.providers import ProviderProfile, ThinkingWire, register_provider
|
||||
from polygateway.thinking import DEFAULT_CAPABILITIES, ThinkingCapability, get_capability
|
||||
from polygateway.types import EFFORT_ORDER, Effort
|
||||
|
||||
@@ -141,6 +142,9 @@ async def _run_rounds(rounds: int, *, stream: bool = True, **source_overrides) -
|
||||
# 不可复核,而 thinking_chars 正是本次改判的直接证据
|
||||
"thinking_observation": resp.thinking_observation,
|
||||
"thinking_chars": len(resp.thinking),
|
||||
# 核对模型身份: 结论依赖"这组数说的是哪个模型"时(L8 的能力表
|
||||
# 对账),渠道串台会把渠道的路由问题记成库的漂移(issue #20)
|
||||
"model_reported": resp.model_reported,
|
||||
"content": resp.content[:60],
|
||||
}
|
||||
)
|
||||
@@ -196,6 +200,25 @@ def _skip_if_unreachable(exc: Exception, matrix_id: str, desc: str):
|
||||
pytest.skip(f"{matrix_id} 源不可用,已记为未覆盖: {str(exc)[:120]}")
|
||||
|
||||
|
||||
async def _rounds_or_skip(matrix_id: str, desc: str, rounds: int, **source_overrides) -> list[dict]:
|
||||
"""`_run_rounds` 加上"源不可用即记为未覆盖"的兜底(本模块 docstring 的纪律)。
|
||||
|
||||
直接调 `_run_rounds` 的代价有两层,2026-09-05 那次 `-m slow` 两样都踩到了:
|
||||
其一外部抖动会以 FAIL 的形态冒出来,与"库真的坏了"无法区分;其二异常发生在
|
||||
`_record()` **之前**,报告里连一行「未覆盖」都不会留下——事后翻报告只看到该
|
||||
矩阵行凭空消失,判断不出当时到底发生了什么。
|
||||
|
||||
只吞网关/网络三类。**不吞 `ValueError` / `RequestRejectedError`**: 前者是装配
|
||||
守卫,后者是"请求本身被拒",两者都是本组要抓的真失败,吞掉即成静默。
|
||||
"""
|
||||
try:
|
||||
return await _run_rounds(rounds, **source_overrides)
|
||||
except (AllSourcesExhausted, SourceDeadError, TransientError) as exc:
|
||||
# `_skip_if_unreachable` 内部 `pytest.skip` 必抛,此处不会落到函数末尾
|
||||
_skip_if_unreachable(exc, matrix_id, desc)
|
||||
raise # pragma: no cover —— 只为让静态读者看清控制流不会往下走
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", autouse=True)
|
||||
def _write_report():
|
||||
yield
|
||||
@@ -244,11 +267,12 @@ class TestMiniMaxM3:
|
||||
"""M3 是唯一实测可关闭推理的 MiniMax 模型,修复的地基压在它身上。"""
|
||||
|
||||
async def test_l1_disable_actually_disables(self):
|
||||
obs = await _run_rounds(_ROUNDS, model="MiniMax-M3", enable_thinking=False)
|
||||
desc = "enable_thinking=False(流式)"
|
||||
obs = await _rounds_or_skip("L1", desc, _ROUNDS, model="MiniMax-M3", enable_thinking=False)
|
||||
offs = [o for o in obs if _reasoning_off(o)]
|
||||
_record(
|
||||
"L1",
|
||||
"enable_thinking=False(流式)",
|
||||
desc,
|
||||
"PASS" if len(offs) == len(obs) else "FAIL",
|
||||
f"{len(offs)}/{len(obs)} 轮未观测到推理",
|
||||
obs,
|
||||
@@ -256,11 +280,12 @@ class TestMiniMaxM3:
|
||||
assert len(offs) == len(obs), f"关闭方向要求每轮满足: {obs}"
|
||||
|
||||
async def test_l2_enable_actually_enables(self):
|
||||
obs = await _run_rounds(_ROUNDS, model="MiniMax-M3", enable_thinking=True)
|
||||
desc = "enable_thinking=True(流式,注入 medium)"
|
||||
obs = await _rounds_or_skip("L2", desc, _ROUNDS, model="MiniMax-M3", enable_thinking=True)
|
||||
ons = [o for o in obs if _reasoning_on(o)]
|
||||
_record(
|
||||
"L2",
|
||||
"enable_thinking=True(流式,注入 medium)",
|
||||
desc,
|
||||
"PASS" if len(ons) * 2 > len(obs) else "FAIL",
|
||||
f"{len(ons)}/{len(obs)} 轮观察到推理",
|
||||
obs,
|
||||
@@ -275,13 +300,14 @@ class TestMiniMaxM3:
|
||||
供应商改模板也不会让它假红。
|
||||
"""
|
||||
rounds = max(3, _ROUNDS // 3)
|
||||
off = await _run_rounds(rounds, model="MiniMax-M3", enable_thinking=False)
|
||||
on = await _run_rounds(rounds, model="MiniMax-M3", enable_thinking=True)
|
||||
desc = "关闭/开启的 prompt_tokens 可分"
|
||||
off = await _rounds_or_skip("L2b", desc, rounds, model="MiniMax-M3", enable_thinking=False)
|
||||
on = await _rounds_or_skip("L2b", desc, rounds, model="MiniMax-M3", enable_thinking=True)
|
||||
off_max = max(o["prompt_tokens"] for o in off)
|
||||
on_min = min(o["prompt_tokens"] for o in on)
|
||||
_record(
|
||||
"L2b",
|
||||
"关闭/开启的 prompt_tokens 可分",
|
||||
desc,
|
||||
"PASS" if off_max < on_min else "FAIL",
|
||||
f"关闭档最大 {off_max} < 开启档最小 {on_min}",
|
||||
off + on,
|
||||
@@ -291,14 +317,15 @@ class TestMiniMaxM3:
|
||||
)
|
||||
|
||||
async def test_l3_no_opinion_is_the_model_default(self):
|
||||
obs = await _run_rounds(_ROUNDS, model="MiniMax-M3", enable_thinking=None)
|
||||
desc = "enable_thinking=None(不干预,基线)"
|
||||
obs = await _rounds_or_skip("L3", desc, _ROUNDS, model="MiniMax-M3", enable_thinking=None)
|
||||
# M3 的默认档实测就是不推理(findings §2.1),所以不干预时也应观测不到推理。
|
||||
# 注意这**不能**反过来证明关闭方向生效 —— L1 与本行同分布,区分二者的是
|
||||
# L2b 的 prompt_tokens 与 L3b 的乱码值反证
|
||||
quiet = [o for o in obs if _reasoning_off(o)]
|
||||
_record(
|
||||
"L3",
|
||||
"enable_thinking=None(不干预,基线)",
|
||||
desc,
|
||||
"PASS" if len(quiet) == len(obs) else "FAIL",
|
||||
f"{len(quiet)}/{len(obs)} 轮未观测到推理(M3 默认档本就不推理)",
|
||||
obs,
|
||||
@@ -323,19 +350,22 @@ class TestMiniMaxM3:
|
||||
上,拿到的会是异常而非"不推理",是假红。
|
||||
"""
|
||||
rounds = max(3, _ROUNDS // 3)
|
||||
bogus = await _run_rounds(
|
||||
desc = "非法值反证 none 被识别"
|
||||
bogus = await _rounds_or_skip(
|
||||
"L3b",
|
||||
desc,
|
||||
rounds,
|
||||
model="MiniMax-M3",
|
||||
enable_thinking=None,
|
||||
extra_body={"reasoning_effort": "definitely-not-a-real-level"},
|
||||
)
|
||||
off = await _run_rounds(rounds, model="MiniMax-M3", enable_thinking=False)
|
||||
off = await _rounds_or_skip("L3b", desc, rounds, model="MiniMax-M3", enable_thinking=False)
|
||||
bogus_on = [o for o in bogus if _reasoning_on(o)]
|
||||
off_quiet = [o for o in off if _reasoning_off(o)]
|
||||
ok = len(bogus_on) * 2 > len(bogus) and len(off_quiet) == len(off)
|
||||
_record(
|
||||
"L3b",
|
||||
"非法值反证 none 被识别",
|
||||
desc,
|
||||
"PASS" if ok else "FAIL",
|
||||
f"非法值 {len(bogus_on)}/{len(bogus)} 轮推理,none {len(off_quiet)}/{len(off)} 轮不推理"
|
||||
"(两者表现不同 ⇒ none 非被丢弃)",
|
||||
@@ -352,7 +382,10 @@ class TestMiniMaxM3:
|
||||
判据是行为而非报文: 若 extra_body 没赢,拿到的就是 none 的结果(不推理)。
|
||||
"""
|
||||
rounds = max(3, _ROUNDS // 2)
|
||||
obs = await _run_rounds(
|
||||
desc = "extra_body 覆盖 profile 注入"
|
||||
obs = await _rounds_or_skip(
|
||||
"L4",
|
||||
desc,
|
||||
rounds,
|
||||
model="MiniMax-M3",
|
||||
enable_thinking=False,
|
||||
@@ -361,7 +394,7 @@ class TestMiniMaxM3:
|
||||
ons = [o for o in obs if _reasoning_on(o)]
|
||||
_record(
|
||||
"L4",
|
||||
"extra_body 覆盖 profile 注入",
|
||||
desc,
|
||||
"PASS" if len(ons) * 2 > len(obs) else "FAIL",
|
||||
f"{len(ons)}/{len(obs)} 轮观察到推理(证明 high 生效而非 none)",
|
||||
obs,
|
||||
@@ -387,8 +420,13 @@ class TestMiniMaxM3:
|
||||
回传正文的余地: 那时裁定会翻成 `OBSERVED`,是好事,不该让它把测试判红。
|
||||
"""
|
||||
rounds = max(3, _ROUNDS // 2)
|
||||
off = await _run_rounds(rounds, stream=False, model="MiniMax-M3", enable_thinking=False)
|
||||
on = await _run_rounds(rounds, stream=False, model="MiniMax-M3", enable_thinking=True)
|
||||
desc = "非流式: prompt 锚点可分 + 开启档如实标 UNKNOWN 而非 ABSENT"
|
||||
off = await _rounds_or_skip(
|
||||
"L5", desc, rounds, stream=False, model="MiniMax-M3", enable_thinking=False
|
||||
)
|
||||
on = await _rounds_or_skip(
|
||||
"L5", desc, rounds, stream=False, model="MiniMax-M3", enable_thinking=True
|
||||
)
|
||||
offs = [o for o in off if _reasoning_off(o)]
|
||||
off_max = max(o["prompt_tokens"] for o in off)
|
||||
on_min = min(o["prompt_tokens"] for o in on)
|
||||
@@ -397,7 +435,7 @@ class TestMiniMaxM3:
|
||||
ok = len(offs) == len(off) and off_max < on_min and len(not_absent) == len(on)
|
||||
_record(
|
||||
"L5",
|
||||
"非流式: prompt 锚点可分 + 开启档如实标 UNKNOWN 而非 ABSENT",
|
||||
desc,
|
||||
"PASS" if ok else "FAIL",
|
||||
f"关闭 {len(offs)}/{len(off)} 轮未观测到推理;"
|
||||
f"关闭档 prompt 最大 {off_max} < 开启档最小 {on_min};"
|
||||
@@ -475,6 +513,16 @@ class TestCapabilityDrift:
|
||||
|
||||
@pytest.mark.parametrize("model", sorted(DEFAULT_CAPABILITIES))
|
||||
async def test_declared_capability_matches_reality(self, model):
|
||||
"""声明 can_disable 的模型必须真的关得掉,否则能力表已漂移。
|
||||
|
||||
**结论依赖模型身份,故先过身份关**: 2026-09-05 实测该渠道对 glm-5 / glm-5.1 /
|
||||
glm-5.2 三个型号的请求全部回报 `model=glm-5.3`(issue #20 的路由问题仍在)。
|
||||
照单全收的话,glm-5.3 那一轮碰巧推理了就会被记成"glm-5 的能力表漂移"——把
|
||||
渠道串台记成库的缺陷,而库这边已经喊了对账告警,行为是对的。
|
||||
身份不符一律 SKIP 记为未覆盖: 那是外部渠道问题,不是能力表的证据。
|
||||
**三个型号一视同仁**,不能只挡报错的那两个: glm-5.2 这次侥幸 PASS(被路由到的
|
||||
glm-5.3 那几轮恰好没推理),而侥幸绿的数据与红的数据一样不可信。
|
||||
"""
|
||||
cap = get_capability(model)
|
||||
provider = _MODEL_PROVIDER[model]
|
||||
rounds = max(3, _ROUNDS // 2)
|
||||
@@ -491,6 +539,16 @@ class TestCapabilityDrift:
|
||||
obs = await _run_rounds(rounds, provider=provider, model=model, enable_thinking=False)
|
||||
except (AllSourcesExhausted, SourceDeadError, TransientError) as exc:
|
||||
_skip_if_unreachable(exc, "L8", desc)
|
||||
strangers = _identity_mismatch(model, obs)
|
||||
if strangers:
|
||||
_record(
|
||||
"L8",
|
||||
desc,
|
||||
"SKIP(身份不符,数据不可信)",
|
||||
f"该渠道把请求回报成 {strangers},本次观测说的不是这个模型",
|
||||
obs,
|
||||
)
|
||||
pytest.skip(f"{model} 被该渠道路由到 {strangers},本次观测说的不是这个模型")
|
||||
offs = [o for o in obs if _reasoning_off(o)]
|
||||
verdict = Counter(_reasoning_off(o) for o in obs)
|
||||
_record(
|
||||
@@ -505,6 +563,20 @@ class TestCapabilityDrift:
|
||||
)
|
||||
|
||||
|
||||
_MYSTERY_PROFILE = ProviderProfile(
|
||||
name="mystery",
|
||||
thinking=ThinkingWire(off=None, on_base=None, effort_key=None),
|
||||
strip_think_tags=False,
|
||||
)
|
||||
"""形态完全未知的 provider(issue #5 守卫的对象),与单元测试 `_MYSTERY` 同款。
|
||||
|
||||
**为什么不再借用默认表里的某一段**: L9 原先拿 `openai` 段当"形态未知"的样本,而
|
||||
1.3.3 起该段已按 OpenAI 标准形态登记(`off={"reasoning_effort":"none"}`、
|
||||
`on_base={}`、`effort_key="reasoning_effort"`),前提消失,用例随之 DID NOT RAISE。
|
||||
守的不变量一天没变,变的只是"哪个段当时恰好没形态"——所以样本改为显式构造,
|
||||
让本条测的是**机制**而不是默认表某一格的当下取值。"""
|
||||
|
||||
|
||||
class TestAssemblyGuardAgainstRealConfig:
|
||||
"""L9: 纯本地,但用的是 .env 里的真实配置形态,防"守卫只在合成配置上生效"。"""
|
||||
|
||||
@@ -516,11 +588,18 @@ class TestAssemblyGuardAgainstRealConfig:
|
||||
_record("L9", "M2.7 + enable_thinking=False", "PASS", "装配期报错,未发出任何请求")
|
||||
|
||||
def test_l9_unknown_shape_rejected_at_assembly(self):
|
||||
"""形态未知的 provider 配了推理开关 → 装配期报错并指路 `register_provider`。
|
||||
|
||||
样本经 `register_provider` 挂进注册表再用,而不是拿默认表里"当时恰好没形态"
|
||||
的那一段——后者的前提会随默认表增补而失效(见 `_MYSTERY_PROFILE`)。
|
||||
"""
|
||||
registry = register_provider(_MYSTERY_PROFILE)
|
||||
with pytest.raises(ValueError, match="register_provider"):
|
||||
GatewayClient.from_settings(
|
||||
_settings(provider="openai", model="kimi-k3", enable_thinking=False)
|
||||
_settings(provider="mystery", model="kimi-k3", enable_thinking=False),
|
||||
registry=registry,
|
||||
)
|
||||
_record("L9", "provider=openai 形态未知", "PASS", "装配期报错并指路")
|
||||
_record("L9", "形态未知的 provider(构造)", "PASS", "装配期报错并指路")
|
||||
|
||||
async def test_transport_layer_rejects_when_guard_is_bypassed(self):
|
||||
"""构造函数全量注入这条路绕过装配守卫,transport 必须兜住并归四分类。"""
|
||||
@@ -674,7 +753,15 @@ async def _probe_effort(
|
||||
# 某个模型在网关上不通时,连续失败会把熔断门打开,后续轮次抛的是
|
||||
# `CircuitOpenError`(同一父类的兄弟)。只捕子类会让"源不可用"这
|
||||
# 件事在第 N 轮换个类型冒出去,把数据采集打断成一次红测
|
||||
return {**base, "error": f"{type(exc).__name__}: {str(exc)[:160]}"}
|
||||
return {
|
||||
**base,
|
||||
"error": f"{type(exc).__name__}: {str(exc)[:160]}",
|
||||
# 另存机器可判的两格: 「上游拒绝这一档」与「该渠道没有这个型号」
|
||||
# 都是 `RequestRejectedError`,`_probe_rejected` 要靠状态码与
|
||||
# 响应体里的 `type` 把它们分开,而不是去模糊匹配整条 message
|
||||
"error_status": exc.status_code,
|
||||
"error_body": exc.body_text,
|
||||
}
|
||||
return {
|
||||
**base,
|
||||
"error": None,
|
||||
@@ -697,7 +784,43 @@ async def _probe_effort(
|
||||
|
||||
|
||||
def _probe_ok(obs: dict) -> bool:
|
||||
return obs["error"] is None
|
||||
"""这一轮拿到了真实观测。
|
||||
|
||||
用 `.get` 而非下标: `_identity_mismatch` 被 L8 复用,而 `_run_rounds` 产出的
|
||||
逐轮字典里根本没有 `error` 键(那条路径上失败是冒泡的,不会留下失败轮)。
|
||||
"""
|
||||
return obs.get("error") is None
|
||||
|
||||
|
||||
def _model_missing(obs: dict) -> bool:
|
||||
"""这一轮失败的原因是**该渠道根本没有这个型号**(404 `model_not_found`)。
|
||||
|
||||
2026-09-05 实测: kimi-for-coding 上午 09:44 四项全 PASS 且 `model_reported`
|
||||
正确,15:00 就变成 `404 | {"error":{...,"type":"model_not_found"}}` —— 渠道
|
||||
把它从账号组里摘掉了。这与"上游拒绝这一档"(400 invalid tier)完全不是一件事:
|
||||
后者是**关于档位的结论**,前者对档位一无所知,只说明源当下不可用。混为一谈会
|
||||
让一次渠道调整变成"能力表漂移"的假红,严重时反过来把能力表改错。
|
||||
|
||||
判据取 `status_code` 与响应体里的 `type` 字段(机器可判的那格),不做整条
|
||||
message 的模糊匹配 —— message 里还拼着源名与库自己的话,匹配它等于赌文案不变。
|
||||
"""
|
||||
return obs.get("error_status") == 404 and "model_not_found" in (obs.get("error_body") or "")
|
||||
|
||||
|
||||
def _probe_rejected(obs: dict) -> bool:
|
||||
"""上游明确拒绝**这一档**(400 / Unsupported value)⇒ 结论: 该档不受支持。
|
||||
|
||||
显式排除 `_model_missing`: 型号不存在时上游没有对档位表过任何态。
|
||||
"""
|
||||
return (obs.get("error") or "").startswith("RequestRejected") and not _model_missing(obs)
|
||||
|
||||
|
||||
def _unreachable_verdict(observations: list[dict]) -> str:
|
||||
"""源不可用的两种成因在报告里必须分得开: 渠道摘了型号 vs 渠道当下抖动。"""
|
||||
broken = [o for o in observations if o.get("error")]
|
||||
if broken and all(_model_missing(o) for o in broken):
|
||||
return "SKIP(源不可用: 该渠道未提供此型号)"
|
||||
return "SKIP(源不可用)"
|
||||
|
||||
|
||||
def _probe_quiet(obs: dict) -> bool:
|
||||
@@ -746,6 +869,9 @@ def _identity_mismatch(model: str, observations: list[dict]) -> list[str]:
|
||||
模型身份的,对不上就必须当场作废,而不是打个折扣继续用。
|
||||
|
||||
`None`(上游未上报)不算不符: 那是"没说",不是"说了别的"。
|
||||
|
||||
定义在 T10 段内但**不专属于它**: L8 的能力表对账同样以模型身份为前提,
|
||||
2026-09-05 那次假红就是它缺了这道关(见该用例 docstring)。
|
||||
"""
|
||||
allowed = {model, *_MODEL_REPORTED_ALIASES.get(model, frozenset())}
|
||||
return sorted(
|
||||
@@ -866,14 +992,17 @@ class TestTierProbe:
|
||||
model, Effort.NONE, rounds=_TIER_ROUNDS, prompt=_TIER_PROMPT, prompt_kind="short"
|
||||
)
|
||||
# 上游拒绝这一档(400)是**结论**而非故障: 它等价于"关不掉";
|
||||
# 其余失败(渠道下线/超时)才是源不可用,按既有纪律记为未覆盖
|
||||
rejected = [o for o in short if o["error"] and o["error"].startswith("RequestRejected")]
|
||||
# 其余失败(渠道下线/型号被摘/超时)才是源不可用,按既有纪律记为未覆盖。
|
||||
# 404 `model_not_found` 因此不算 rejected —— 它会自然落进下面那条源不可用分支
|
||||
rejected = [o for o in short if _probe_rejected(o)]
|
||||
usable = [o for o in short if _probe_ok(o)]
|
||||
# **按可用轮判,而不是一有失败就整条跳过**: 共用网关上偶发 429/503 是常态,
|
||||
# 一票否决会让整张表因为一次抖动而没有数据。样本低于 3 轮才是真的没结论
|
||||
if not rejected and len(usable) < min(3, _TIER_ROUNDS):
|
||||
broken = [o for o in short if o["error"]]
|
||||
_probe_record(model, "none 方向", "SKIP(源不可用)", _rt_summary(short), short)
|
||||
_probe_record(
|
||||
model, "none 方向", _unreachable_verdict(short), _rt_summary(short), short
|
||||
)
|
||||
pytest.skip(f"{model} 源不可用,已记为未覆盖: {broken[0]['error'][:120]}")
|
||||
|
||||
strangers = _identity_mismatch(model, short)
|
||||
@@ -959,13 +1088,14 @@ class TestTierProbe:
|
||||
if not tiers:
|
||||
pytest.skip(f"{model} 只登记了 none,没有开启档可验")
|
||||
failures = []
|
||||
verdicts = []
|
||||
for tier in tiers:
|
||||
observations = await _probe_effort(
|
||||
model, tier, rounds=_TIER_ROUNDS, prompt=_TIER_PROMPT, prompt_kind="short"
|
||||
)
|
||||
rejected = [
|
||||
o for o in observations if o["error"] and o["error"].startswith("RequestRejected")
|
||||
]
|
||||
# 同 none 方向: 只有"拒绝这一档"才是关于档位的结论,404 型号不存在
|
||||
# 说明的是源不可用,记成 FAIL 会把渠道摘型号读成"登记了个上游不认的档"
|
||||
rejected = [o for o in observations if _probe_rejected(o)]
|
||||
usable = [o for o in observations if _probe_ok(o)]
|
||||
observed = [o for o in observations if _probe_observed(o)]
|
||||
strangers = _identity_mismatch(model, observations)
|
||||
@@ -982,17 +1112,22 @@ class TestTierProbe:
|
||||
if rejected:
|
||||
verdict, problem = "FAIL(上游拒绝该档)", f"{tier.value}: 上游拒绝"
|
||||
elif not usable:
|
||||
verdict, problem = "SKIP(源不可用)", None
|
||||
verdict, problem = _unreachable_verdict(observations), None
|
||||
elif len(observed) * 2 > len(usable):
|
||||
verdict, problem = "PASS", None
|
||||
else:
|
||||
verdict, problem = "FAIL(该档未推理)", f"{tier.value}: 多数轮未观测到推理"
|
||||
if problem:
|
||||
failures.append(problem)
|
||||
verdicts.append(verdict)
|
||||
_probe_record(
|
||||
model, f"档位 {tier.value}", verdict, _rt_summary(observations), observations
|
||||
)
|
||||
assert not failures, f"{model} 登记的档位与实测不符: {failures}"
|
||||
if all(v.startswith("SKIP") for v in verdicts):
|
||||
# 一档都没跑通却判绿,就是本模块 docstring 明令禁止的"静默计入通过":
|
||||
# 绿色在这里会被读成"登记的档位都验过了",而实情是一条都没验
|
||||
pytest.skip(f"{model} 各档均源不可用,已记为未覆盖: {verdicts}")
|
||||
|
||||
@pytest.mark.parametrize("model", ["gemini-3.1-pro", "gpt-5.5", "glm-5.3"])
|
||||
async def test_t10_no_opinion_stays_no_opinion(self, model):
|
||||
@@ -1031,6 +1166,10 @@ class TestTierProbe:
|
||||
"effort": "(不表态)",
|
||||
"prompt_kind": "short",
|
||||
"error": f"{type(exc).__name__}: {str(exc)[:160]}",
|
||||
# 与 `_probe_effort` 的失败轮同形: 少这两格,
|
||||
# `_unreachable_verdict` 会把"型号被摘"读成普通抖动
|
||||
"error_status": exc.status_code,
|
||||
"error_body": exc.body_text,
|
||||
}
|
||||
)
|
||||
continue
|
||||
@@ -1057,7 +1196,7 @@ class TestTierProbe:
|
||||
_probe_record(
|
||||
model,
|
||||
"默认档基线(不表态)",
|
||||
"SKIP(源不可用)",
|
||||
_unreachable_verdict(observations),
|
||||
_rt_summary(observations),
|
||||
observations,
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user