"""推理裁定与对账的行为测试(issue #16/#17 设计 §4-§5)。 判据来自 2026-08-25 实测(findings): MiniMax-M3 在开启档流式路径下返回 185 字符 推理正文却不上报 `completion_tokens_details`,而 qwen/deepseek 两者都报。库因此 不能把任何单一信号当权威——本组用例逐条钉死"哪个信号该赢"。 """ import pytest from loguru import logger from polygateway.providers import ProviderProfile, ThinkingWire, get_provider from polygateway.thinking import ( DEFAULT_CAPABILITIES, ThinkingCapability, get_capability, observe_thinking, reconcile_thinking, register_capability, resolve_thinking, ) from polygateway.types import Effort, ThinkingObservation _MYSTERY = ProviderProfile( name="mystery", thinking=ThinkingWire(off=None, on_base=None, effort_key=None), strip_think_tags=False, ) """形态完全未知的 provider(issue #5 的守卫对象)。 2026-09-04 起默认表 8 段全部有形态,故未知样本改为显式构造——测的是**机制** (不知道怎么表达就报错并指路),不是某个段当时的配置。""" def _warnings(): """捕获库发出的 WARNING;loguru 不经标准 logging,pytest 的 caplog 抓不到。""" messages: list[str] = [] sink_id = logger.add(messages.append, level="WARNING") return messages, sink_id class TestObserveThinking: """三态裁定: 证据硬度决定优先级,无信号一律 UNKNOWN。""" def test_reasoning_text_alone_proves_it_happened(self): """推理正文是事实本身: 上游不报 token 数也照样成立(M3 流式实测形态)。""" assert ( observe_thinking(thinking="先解方程 x+y=35", reasoning_tokens=None) is ThinkingObservation.OBSERVED ) def test_blank_text_is_not_evidence(self): """纯空白正文不算证据: 网关响应是外部输入,truthy 判据会把空格计成推理(P5)。""" assert ( observe_thinking(thinking=" \n\t ", reasoning_tokens=None) is ThinkingObservation.UNKNOWN ) def test_positive_token_count_proves_it_happened(self): """无正文但上游报了推理用量(qwen 非流式形态)。""" assert observe_thinking(thinking="", reasoning_tokens=205) is ThinkingObservation.OBSERVED def test_zero_token_count_is_positive_evidence_of_absence(self): """`0` 是"上报了且为零",与"没上报"语义不同,故是 ABSENT 而非 UNKNOWN。""" assert observe_thinking(thinking="", reasoning_tokens=0) is ThinkingObservation.ABSENT def test_no_signal_at_all_stays_unknown(self): """M3 非流式开启档的真实形态: 推理已计费却既无正文也无 token 数。 判成 ABSENT 就是伪装成"没推理"——正是 issue #16/#17 的病根。 """ assert observe_thinking(thinking="", reasoning_tokens=None) is ThinkingObservation.UNKNOWN def test_text_outranks_a_zero_count(self): """转述与事实冲突时事实赢: 正文在,`reasoning_tokens=0` 不能翻案。""" assert ( observe_thinking(thinking="想了想", reasoning_tokens=0) is ThinkingObservation.OBSERVED ) @pytest.mark.parametrize("negative", [-1, -205]) def test_negative_token_count_is_not_evidence_of_absence(self, negative): """负数是坏数据,不是"上游明确上报未推理"这个最强的正面结论。 当前 transport 已在边界把负数归 `None`,所以这条走不通;但本函数的 docstring 自称"外部输入校验后使用",第二个 transport 直接填该值时, `> 0 else ABSENT` 会给出一个方向相反的强结论。函数自身必须闭合(P5)。 """ assert observe_thinking(thinking="", reasoning_tokens=negative) is ( ThinkingObservation.UNKNOWN ) class TestThinkingObservationEnum: def test_values_are_stable_strings(self): """取值进遥测落库,改名即历史数据断层。""" assert ThinkingObservation.OBSERVED == "observed" assert ThinkingObservation.ABSENT == "absent" assert ThinkingObservation.UNKNOWN == "unknown" def test_enum_lives_in_the_innermost_layer(self): """枚举必须定义在 `types.py`(最内层)。 它是 `LLMResponse` 的字段类型;定义在决策层 `thinking.py` 会让 `types.py` 反向 import 决策模块,违反 P7 依赖铁律(import-linter 契约执法)。 """ assert ThinkingObservation.__module__ == "polygateway.types" @pytest.mark.parametrize("bogus", ["", "OBSERVED", "yes", "none"]) def test_unknown_strings_are_rejected(bogus): """非法值必须抛 ValueError: 缓存回放与遥测归一化都靠它识别域外取值(设计 §6)。 两处接住这个 ValueError 后**降级而非作废**(缓存复活内容 + 记 UNKNOWN、遥测 照常落行),但降级的前提是构造器真的会拒绝——它一旦放行,域外取值就会一路 进到 `LLMResponse` 与遥测列里。 """ with pytest.raises(ValueError): ThinkingObservation(bogus) class TestThinkingCapability: """issue #5: 能力按 model 登记——同一 provider 内部代际差异是决定性的。""" def test_registered_models_carry_evidence(self): """登记必须附实测证据: 表会过期,没有出处就无从判断该不该信。""" for model in ("MiniMax-M3", "MiniMax-M2.7", "MiniMax-M2.5"): cap = get_capability(model) assert cap is not None and cap.evidence.strip() def test_m3_can_disable_but_m2x_cannot(self): assert get_capability("MiniMax-M3").can_disable is True assert get_capability("MiniMax-M2.7").can_disable is False assert get_capability("MiniMax-M2.5").can_disable is False def test_unregistered_model_is_unknown(self): assert get_capability("some-brand-new-model") is None def test_register_capability_is_pure(self): table = register_capability("x-1", ThinkingCapability((Effort.NONE, Effort.AUTO), "实测")) assert get_capability("x-1", table=table) is not None assert get_capability("x-1") is None # 默认表未被污染 def test_default_capabilities_mapping_is_read_only(self): with pytest.raises(TypeError): DEFAULT_CAPABILITIES["hack"] = None # type: ignore[index] class TestResolveThinking: """五条判定规则(顺序即语义);设计 §5 真值表。""" def test_rule1_none_injects_nothing(self): got = resolve_thinking(get_provider("minimax"), None, None, model="MiniMax-M3") assert got == {} @pytest.mark.parametrize("enable", [True, False]) def test_rule2_unknown_shape_raises_and_points_the_way(self, enable): with pytest.raises(ValueError, match="register_provider") as exc: resolve_thinking(_MYSTERY, None, enable, model="kimi-k3") assert "extra_body" in str(exc.value) def test_rule3_unregistered_model_warns_but_passes(self): messages, sink_id = _warnings() try: got = resolve_thinking(get_provider("minimax"), None, False, model="MiniMax-M9") finally: logger.remove(sink_id) assert got == {"reasoning_effort": "none"} assert any("MiniMax-M9" in m for m in messages) def test_rule4_cannot_disable_raises_with_the_model_name(self): cap = get_capability("MiniMax-M2.7") with pytest.raises(ValueError, match="MiniMax-M2.7"): resolve_thinking(get_provider("minimax"), cap, False, model="MiniMax-M2.7") def test_rule4_only_blocks_the_off_direction(self): """关不掉 ≠ 开不了: M2.x 默认就在推理,开的方向不该被拦。""" cap = get_capability("MiniMax-M2.7") got = resolve_thinking(get_provider("minimax"), cap, True, model="MiniMax-M2.7") # 2026-09-04: "开"不再硬编码 medium——那是替下游做的档位判断,且 medium 在 # GLM/kimi/deepseek 的档位表里根本不存在。MiniMax 开启档本就无需参数 assert got == {} def test_rule5_normal_path(self): cap = get_capability("MiniMax-M3") assert resolve_thinking(get_provider("minimax"), cap, False, model="MiniMax-M3") == { "reasoning_effort": "none" } def test_unknown_shape_beats_capability_check(self): """第 2 步先于第 4 步: 形态未知时无从注入,能力如何无关紧要。""" cap = ThinkingCapability((Effort.AUTO,), "构造") with pytest.raises(ValueError, match="register_provider"): resolve_thinking(_MYSTERY, cap, False, model="whatever") class TestReconcileThinking: """声明 × 观测对账(设计 §5): 矛盾出文案,不表态出 None。 文案本身是被断言对象——判定与日志分离正是为此: 告警内容可直接比对,不必 去解析日志格式。 """ _CAP = ThinkingCapability( (Effort.NONE, Effort.AUTO), "2026-08-02 实测 reasoning_effort=none 可关闭" ) def test_off_but_observed_with_a_registered_capability_blames_the_table(self): """已登记却实测推理了 = 能力表漂移: 必须附 evidence 与更新指路。""" msg = reconcile_thinking( enable_thinking=False, observation=ThinkingObservation.OBSERVED, capability=self._CAP, model="MiniMax-M3", ) assert msg is not None assert "MiniMax-M3" in msg assert "2026-08-02 实测 reasoning_effort=none 可关闭" in msg assert "register_capability" in msg def test_off_but_observed_unregistered_never_claims_a_table_entry(self): """未登记模型没有"能力表声称"这回事——说它就是撒谎。""" msg = reconcile_thinking( enable_thinking=False, observation=ThinkingObservation.OBSERVED, capability=None, model="MiniMax-M9", ) assert msg is not None assert "MiniMax-M9" in msg assert "能力表" not in msg assert "register_capability" in msg def test_registered_and_unregistered_wordings_differ(self): registered = reconcile_thinking( enable_thinking=False, observation=ThinkingObservation.OBSERVED, capability=self._CAP, model="MiniMax-M3", ) unregistered = reconcile_thinking( enable_thinking=False, observation=ThinkingObservation.OBSERVED, capability=None, model="MiniMax-M3", ) assert registered != unregistered @pytest.mark.parametrize("capability", [None, _CAP]) def test_on_but_absent_is_a_contradiction(self, capability): """上游明确上报未推理: 这是唯一的正面证伪,与能力表登记与否无关。""" msg = reconcile_thinking( enable_thinking=True, observation=ThinkingObservation.ABSENT, capability=capability, model="qwen3.7-plus", ) assert msg is not None assert "qwen3.7-plus" in msg @pytest.mark.parametrize("capability", [None, _CAP]) def test_on_but_unknown_admits_it_cannot_confirm(self, capability): """issue #17 的诚实版本: 明说"我注入了,但我看不见结果"。""" msg = reconcile_thinking( enable_thinking=True, observation=ThinkingObservation.UNKNOWN, capability=capability, model="MiniMax-M3", ) assert msg is not None assert "MiniMax-M3" in msg def test_off_and_absent_stays_silent(self): """要求关闭 + 上游明确上报未推理 = 要求被满足,没有可报的矛盾。 这一格与 `test_off_and_unknown_stays_silent` 的沉默理由**不同**: 那里是 "没有证伪力",这里是"正面证实要求已满足"。两者都必须沉默,漏测哪一格, 把 Phase 2 的判据写成 `is ABSENT` 之类的反向条件都不会被抓住。 """ assert ( reconcile_thinking( enable_thinking=False, observation=ThinkingObservation.ABSENT, capability=self._CAP, model="qwen3.7-plus", ) is None ) def test_off_and_unknown_stays_silent(self): """UNKNOWN 没有证伪力: 拿它报警等于每次关闭调用都喊(M3 关闭档恒落此档)。""" assert ( reconcile_thinking( enable_thinking=False, observation=ThinkingObservation.UNKNOWN, capability=self._CAP, model="MiniMax-M3", ) is None ) @pytest.mark.parametrize( "observation", [ThinkingObservation.OBSERVED, ThinkingObservation.ABSENT, ThinkingObservation.UNKNOWN], ) def test_no_request_no_grievance(self, observation): """调用方不表态,就无从谈"违背"。""" assert ( reconcile_thinking( enable_thinking=None, observation=observation, capability=self._CAP, model="MiniMax-M3", ) is None ) def test_on_and_observed_is_exactly_what_was_asked_for(self): assert ( reconcile_thinking( enable_thinking=True, observation=ThinkingObservation.OBSERVED, capability=self._CAP, model="MiniMax-M3", ) is None ) class TestEffortVocabulary: """八档封闭词汇(设计 §3.1);`auto` 不可省——9 个纯开关型模型无强度档可填。""" def test_none_and_auto_are_distinct_members(self): assert Effort.NONE != Effort.AUTO assert Effort("none") is Effort.NONE assert Effort("auto") is Effort.AUTO def test_vocabulary_is_exactly_eight(self): assert len(list(Effort)) == 8 def test_values_are_wire_literals(self): # 档位值直接写进请求体,改名即改变发出去的字节 assert [e.value for e in Effort] == [ "none", "auto", "minimal", "low", "medium", "high", "xhigh", "max", ] class TestCapabilityTierList: """能力表从 bool 变成档位清单(设计 §3.2);三个派生量不存字段,存了必漂移。""" def test_capability_derives_can_disable(self): assert ThinkingCapability((Effort.NONE, Effort.AUTO), "实测").can_disable is True assert ThinkingCapability((Effort.LOW, Effort.MAX), "实测").can_disable is False def test_cheapest_effort_skips_none(self): # 「关不掉时的可执行替代」取的是除 none 外最弱的一档 assert ( ThinkingCapability((Effort.LOW, Effort.HIGH, Effort.MAX), "实测").cheapest_effort is Effort.LOW ) assert ( ThinkingCapability((Effort.NONE, Effort.HIGH, Effort.MAX), "实测").cheapest_effort is Effort.HIGH ) assert ThinkingCapability((Effort.NONE, Effort.AUTO), "实测").cheapest_effort is Effort.AUTO assert ThinkingCapability((Effort.AUTO,), "实测").cheapest_effort is Effort.AUTO def test_cheapest_effort_is_none_when_only_none(self): # 只能关不能开: 没有可推荐的「最省的开启档」 assert ThinkingCapability((Effort.NONE,), "实测").cheapest_effort is None def test_is_tiered_excludes_none_and_auto(self): # 纯开关型模型不该被告知「可选档位」——它没有档位 assert ThinkingCapability((Effort.NONE, Effort.AUTO), "实测").is_tiered is False assert ThinkingCapability((Effort.AUTO,), "实测").is_tiered is False assert ThinkingCapability((Effort.LOW, Effort.MAX), "实测").is_tiered is True def test_empty_efforts_rejected(self): with pytest.raises(ValueError, match="至少"): ThinkingCapability((), "实测") def test_duplicate_efforts_rejected(self): with pytest.raises(ValueError, match="重复"): ThinkingCapability((Effort.LOW, Effort.LOW), "实测") def test_glm53_cannot_be_disabled(self): # 三源一致(智谱官方文档/cherry-studio/OpenRouter): thinking.type 只接受 enabled cap = get_capability("glm-5.3") assert cap is not None assert cap.can_disable is False assert cap.cheapest_effort is Effort.LOW def test_m2_series_still_cannot_be_disabled(self): # 迁移回归: 旧表用 can_disable=False 表达的事实,新表用「none 不在清单里」表达 assert get_capability("MiniMax-M2.7").can_disable is False assert get_capability("MiniMax-M2.5").can_disable is False assert get_capability("MiniMax-M3").can_disable is True