feat: make the thinking switch real and collect reasoning tokens (issue #5, #6)

enable_thinking=False was a silent no-op for minimax and openai sources.
The shape of the switch now stays at provider level while a model-level
capability table says whether a given model can honour it at all, and
reasoning_tokens is collected so the cost of thinking can be told apart
from the cost of answering.

Verified against the live gateway: a seventeen-row matrix over 137 real
calls, kept out of the CI gate behind the slow marker.
This commit is contained in:
2026-08-02 08:14:52 -04:00
30 changed files with 1917 additions and 56 deletions
+1 -1
View File
@@ -31,7 +31,7 @@ from polygateway.types import (
SourceConfig,
)
__version__ = "1.0.5"
__version__ = "1.0.6"
__all__ = [
"DEFAULT_PROFILES",
+46 -12
View File
@@ -25,7 +25,7 @@ from polygateway.middleware.retry import RetryMW
from polygateway.middleware.structured import StructuredMW
from polygateway.middleware.telemetry import TelemetryEmitter, TelemetryMW
from polygateway.pricing import PricingTable
from polygateway.providers import get_provider
from polygateway.providers import get_capability, get_provider, resolve_thinking
from polygateway.sources import (
AdaptivePacer,
HealthAwareSelector,
@@ -51,7 +51,7 @@ if TYPE_CHECKING:
TelemetryRecorder,
Transport,
)
from polygateway.providers import ProviderProfile
from polygateway.providers import ProviderProfile, ThinkingCapability
from polygateway.types import (
BackpressurePolicy,
RetryPolicy,
@@ -61,22 +61,52 @@ if TYPE_CHECKING:
_T = TypeVar("_T")
def _guard_thinking(
sources: list[SourceConfig],
profiles: list[ProviderProfile],
capabilities: Mapping[str, ThinkingCapability] | None,
) -> None:
"""装配期把不可满足的推理开关炸掉,而不是留到运行时(issue #5)。
与 transport 内的同一次判定不是重复: 那里兜的是"构造函数全量注入"这条路
(CLAUDE.md §4.5 的第二条装配路),而工厂路占 90% 场景,配置错误应当在装配期
就带着指路信息炸掉。`get_provider` 现在就是同一形态的双点调用。
"""
for source, profile in zip(sources, profiles, strict=True):
resolve_thinking(
profile,
get_capability(source.model, table=capabilities),
source.enable_thinking,
model=source.model,
)
def _fingerprint_mark(source: SourceConfig) -> str:
"""单源的指纹标记;`enable_thinking` 仅在**表态时**追加。
只在表态时追加不是省事: 这样只配了 `extra_body` 的存量源字面量与 issue #4
时期逐字相同,升级本版本不会给它们平白来一次全量缓存冷启动。
"""
parts: list[Any] = [source.model, dict(source.extra_body)]
if source.enable_thinking is not None:
parts.append(source.enable_thinking)
return json.dumps(parts, sort_keys=True, ensure_ascii=False)
def build_model_fingerprint(sources: Iterable[SourceConfig]) -> str:
"""缓存 key 的模型身份: 多源 scope = 排序去重的 model 合集。
配置级采样参数(`extra_body`)必须参与,否则把 temperature 从 0 改成 1
后重启仍会读到旧缓存(issue #4 设计决策 C)。全源 `extra_body` 皆空时
字面量与历史实现逐字相同,不触发存量缓存冷启动。
后重启仍会读到旧缓存(issue #4 设计决策 C)。`enable_thinking` 同理
(issue #5): 它一旦真正改变请求体,"关掉推理后重启"就会读到开着推理时
缓存的旧响应。全源两者皆未表态时字面量与历史实现逐字相同,不触发存量
缓存冷启动。
"""
fingerprint = ",".join(sorted({s.model for s in sources}))
# 按 (model, extra_body) 而非源名摘要: 语义是"本 scope 会用哪些
# (模型, 解码参数)组合",改源名不该误触全量冷启动
# 按 (model, extra_body[, enable_thinking]) 而非源名摘要: 语义是"本 scope
# 会用哪些(模型, 请求形态)组合",改源名不该误触全量冷启动
marks = sorted(
{
json.dumps([s.model, dict(s.extra_body)], sort_keys=True, ensure_ascii=False)
for s in sources
if s.extra_body
}
{_fingerprint_mark(s) for s in sources if s.extra_body or s.enable_thinking is not None}
)
if marks:
digest = hashlib.sha256("".join(marks).encode("utf-8")).hexdigest()
@@ -241,11 +271,13 @@ class GatewayClient:
cache: CacheBackend | None = None,
telemetry: TelemetryRecorder | None = None,
registry: Mapping[str, ProviderProfile] | None = None,
capabilities: Mapping[str, ThinkingCapability] | None = None,
rng: Any = random.random,
) -> GatewayClient:
"""按配置装配;显式传入的后端实例即共享(None 项按配置自建私有实例)。"""
sources = list(settings.sources)
profiles = [get_provider(s.provider, registry=registry) for s in sources]
_guard_thinking(sources, profiles, capabilities)
strategy, escalation = _build_structured(profiles)
return cls(
scope=settings.scope,
@@ -253,7 +285,7 @@ class GatewayClient:
selector=_build_selector(settings.selector, rng=rng),
limiter=limiter or _build_limiter(settings, sources),
breaker=breaker or _build_breaker(settings),
transport=OpenAICompatTransport(registry=registry),
transport=OpenAICompatTransport(registry=registry, capabilities=capabilities),
retry=settings.retry,
backpressure=settings.backpressure,
quota_full=settings.quota_full,
@@ -279,6 +311,7 @@ class GatewayClient:
cache: CacheBackend | None = None,
telemetry: TelemetryRecorder | None = None,
registry: Mapping[str, ProviderProfile] | None = None,
capabilities: Mapping[str, ThinkingCapability] | None = None,
env: Mapping[str, str] | None = None,
) -> GatewayClient:
"""从 .env/环境变量装配一个 scope 的 client(键名清单见 .env.example)。"""
@@ -289,6 +322,7 @@ class GatewayClient:
cache=cache,
telemetry=telemetry,
registry=registry,
capabilities=capabilities,
)
+1
View File
@@ -438,6 +438,7 @@ class RetryMW:
usage_source=result.usage_source,
cached_prompt_tokens=result.cached_prompt_tokens,
model_reported=result.model_reported,
reasoning_tokens=result.reasoning_tokens,
)
async def _settle_and_release(self, permit: Permit, actual: int) -> None:
+5
View File
@@ -64,6 +64,7 @@ class TelemetryEmitter:
error=error,
cached_prompt_tokens=response.cached_prompt_tokens if response else None,
model_reported=response.model_reported if response else None,
reasoning_tokens=response.reasoning_tokens if response else None,
# 唯一有"生效源"的入口,故是唯一能并上 extra_body 的(设计决策 D)
sampling=canonical_sampling_json(merge_sampling(source.extra_body, request.sampling)),
)
@@ -90,6 +91,7 @@ class TelemetryEmitter:
# 统计供应商缓存命中率必须带 WHERE cache_hit = false,否则重复计数。
cached_prompt_tokens=response.cached_prompt_tokens,
model_reported=response.model_reported,
reasoning_tokens=response.reasoning_tokens,
# 由最外层 TelemetryMW 调用,手上没有 source。缓存命中行无损:
# sampling 已进缓存 key,能命中即意味调用级参数与历史那次逐字相同
sampling=canonical_sampling_json(request.sampling),
@@ -117,6 +119,7 @@ class TelemetryEmitter:
error=error,
cached_prompt_tokens=None,
model_reported=None,
reasoning_tokens=None,
# 无具体源,与 model/provider/source_name 置空同一先例(设计决策 D)
sampling=canonical_sampling_json(request.sampling),
)
@@ -142,6 +145,7 @@ class TelemetryEmitter:
cached_prompt_tokens: int | None,
model_reported: str | None,
sampling: str | None,
reasoning_tokens: int | None,
) -> None:
try:
# 成本换算(M2 §6): 成功行按单价换算;缓存命中 0.0(未产生新调用);
@@ -182,6 +186,7 @@ class TelemetryEmitter:
cached_prompt_tokens=cached_prompt_tokens,
model_reported=model_reported,
sampling=sampling,
reasoning_tokens=reasoning_tokens,
)
except asyncio.CancelledError:
raise
+1
View File
@@ -275,4 +275,5 @@ class TelemetryRecorder(Protocol):
cached_prompt_tokens: int | None,
model_reported: str | None,
sampling: str | None,
reasoning_tokens: int | None,
) -> None: ...
+174 -16
View File
@@ -10,24 +10,37 @@ from dataclasses import dataclass
from types import MappingProxyType
from typing import Any
from loguru import logger
@dataclass(frozen=True)
class ProviderProfile:
"""单个 provider 的能力与差异声明。
thinking_on/thinking_off 分别是 `SourceConfig.enable_thinking` 为
True/False 时并入请求体的参数片段(None 时二者都不注入,用模型默认);
strip_think_tags 声明响应 content 需剥离 ``<think>`` 标签(qwen 系);
supports_native_schema 供 D14 阶梯选择原生 response_format 策略
True/False 时并入请求体的参数片段(`enable_thinking` 为 None 时二者都不
注入,用模型默认);strip_think_tags 声明响应 content 需剥离 ``<think>``
标签(qwen 系);supports_native_schema 供 D14 阶梯选择原生 response_format。
注: 某个 provider 的两档若皆为空字典(如 openai/minimax),说明该 provider
无已知的推理开关参数——此时 `enable_thinking` 对它**不产生任何效果**,
而非静默生效。需要下发自定义参数时用 `SourceConfig.extra_body`。
两档各有三种取值,**语义互不重叠**(issue #5):
========== ==========================================================
``{...}`` 已知的注入片段
``{}`` 已知**无需注入**任何参数即处于该档
``None`` **未知**: 本库不知道该 provider 如何表达这一档
========== ==========================================================
`None` 与 `{}` 必须分开: 二者曾同为空字典,导致 `enable_thinking=False`
对 minimax/openai 源静默失效——调用方以为关掉了推理,实际什么都没发生。
现在 `None` 会在装配期显式报错并指路 `register_provider` / `extra_body`。
注: 本类只声明**形态**(参数长什么样,按 provider 变);某个具体模型能否
关闭推理属**能力**(按 model 变),见 `ThinkingCapability`。
"""
name: str
thinking_on: dict[str, Any]
thinking_off: dict[str, Any]
thinking_on: Mapping[str, Any] | None
thinking_off: Mapping[str, Any] | None
strip_think_tags: bool
supports_native_schema: bool = False
@@ -47,26 +60,171 @@ DEFAULT_PROFILES: Mapping[str, ProviderProfile] = MappingProxyType(
thinking_off={"thinking": {"type": "disabled"}},
strip_think_tags=False,
),
# 两档皆空 ⇒ `enable_thinking` 对本 provider **不产生任何效果**(调用方
# 以为关掉了实际没关)。真需要控制推理时经 `SourceConfig.extra_body` 下发
# OpenAI 兼容基线段名: 实践中被复用为**任意**兼容厂商的兜底(下游把
# kimi-k3 挂在 provider=openai 下),故不能下发任何厂商方言参数——发给
# 不认识它的厂商会 400。两档标 None(未知): 配了 enable_thinking 即在
# 装配期报错并指路,真 OpenAI 推理模型的用户走 register_provider
"openai": ProviderProfile(
name="openai",
thinking_on={},
thinking_off={},
thinking_on=None,
thinking_off=None,
strip_think_tags=False,
),
# OpenAI 兼容基线,无已知注入差异;reasoning_content 由 transport 通用处理。
# 同上: 两档皆空 ⇒ `enable_thinking` 对 MiniMax 源不产生任何效果
# 注入形态出处: 2026-08-02 经自建 new-api 中转实测(findings §2),
# **直连官方端点未验证**。实测 enable_thinking / thinking 两种写法均被
# 静默丢弃(prompt_tokens 恒定不变),reasoning_effort 才是真开关。
# "开"取 medium: qwen 的 enable_thinking:true 与 deepseek 的
# thinking:{enabled} 都不指定预算、由模型自定,medium 是五档里语义最接近
# "厂商正常强度"的一档;取 high 等于替下游做"加钱换质量"的业务判断。
# 要精确控制档位经 `SourceConfig.extra_body`(优先级高于本片段)
"minimax": ProviderProfile(
name="minimax",
thinking_on={},
thinking_off={},
thinking_on={"reasoning_effort": "medium"},
thinking_off={"reasoning_effort": "none"},
strip_think_tags=False,
),
}
)
class ThinkingUnsupportedError(ValueError):
"""推理开关无法满足: 形态未知或该模型不支持该方向(issue #5)。
是 `ValueError` 的子类而非 `errors.py` 四分类之一——它描述的是**配置**
不可满足(装配期就该炸),不是一次调用的运行时失败。transport 在请求期
捕获它并翻译为 `RequestRejectedError` 再进四分类。单列一个类型是为了让
捕获点能精确到它,而不是宽catch 整个 `ValueError`(那会把序列化等无关
错误误贴成"推理开关无法满足")。
"""
@dataclass(frozen=True)
class ThinkingCapability:
"""某个**具体模型**能否关闭推理(issue #5);登记必须附实测证据与日期。
与 `ProviderProfile` 的分工: 后者声明**形态**(参数长什么样,按 provider 变,
数年不变一次),本类声明**能力**(按 model 变,同一 provider 每代都变)。二者
合一在 provider 级表达不了代际差异——实测 MiniMax-M3 可关闭推理,而同厂的
M2.7/M2.5 三种参数形态全部无效(findings §2.3),profile 一格管不住三个模型。
`evidence` 不是装饰: 能力表过期是必然事件,没有出处就无从判断该不该信它。
"""
can_disable: bool
evidence: str
DEFAULT_CAPABILITIES: Mapping[str, ThinkingCapability] = MappingProxyType(
{
"MiniMax-M3": ThinkingCapability(
can_disable=True,
evidence="2026-08-02 经 new-api 中转实测 N=10: reasoning_effort=none 稳定关闭,零跳变",
),
"MiniMax-M2.7": ThinkingCapability(
can_disable=False,
evidence=(
"2026-08-02 实测 reasoning_effort=none / thinking:{disabled} / thinking:{adaptive} "
"各 N=3 全部无效;OpenRouter 注册表登记 mandatory:true,models.dev 登记无控制手段"
),
),
"MiniMax-M2.5": ThinkingCapability(
can_disable=False,
evidence="2026-08-02 实测同 M2.7: 三种形态各 N=3 全部无效;外部注册表同样登记为强制推理",
),
"qwen3.7-plus": ThinkingCapability(
can_disable=True,
evidence="2026-08-02 实测 enable_thinking=false 关闭(completion 5 token,无推理)",
),
"deepseek-v4-pro": ThinkingCapability(
can_disable=True,
evidence="2026-08-02 实测 thinking:{type:disabled} 关闭(completion 3 token,无推理)",
),
}
)
"""在用模型的推理能力登记(YAGNI: 不覆盖全世界,未登记走 `resolve_thinking` 退化)。"""
def get_capability(
model: str, *, table: Mapping[str, ThinkingCapability] | None = None
) -> ThinkingCapability | None:
"""按模型名精确查找;未登记返回 None(= 能力未知,由调用方决定如何退化)。
与 `get_provider` 未注册即报错不同: provider 是配置里写死的少数几个值,
写错就是配置错误;而模型名千变万化,新模型上线不该被库挡住(设计 §5 R4)。
"""
return (DEFAULT_CAPABILITIES if table is None else table).get(model)
def register_capability(
model: str,
capability: ThinkingCapability,
*,
base: Mapping[str, ThinkingCapability] | None = None,
) -> dict[str, ThinkingCapability]:
"""纯函数注册: 返回 base(缺省 DEFAULT_CAPABILITIES)+ 新条目的新表,同名覆盖。"""
table = dict(DEFAULT_CAPABILITIES if base is None else base)
table[model] = capability
return table
def resolve_thinking(
profile: ProviderProfile,
capability: ThinkingCapability | None,
enable_thinking: bool | None,
*,
model: str,
warn_unregistered: bool = True,
) -> Mapping[str, Any]:
"""三态 + 两层能力 → 请求体注入片段;不可满足时 ValueError。
调用点负责翻译: 装配期直接冒泡(配置错误),transport 内翻译为
`RequestRejectedError`(四分类之一)。判定顺序即语义,不可调换——形态未知时
无从注入,能力如何无关紧要,故 Phase 2 必须先于 Phase 4;未登记模型没有
`can_disable` 可读,故 Phase 3 必须先于 Phase 4。
`model` 只用于错误与告警文案: 报错能定位到具体模型才有可操作性,而
`capability` 为 None(未登记)时无从从别处取得模型名。
`warn_unregistered=False` 供请求热路径去重用: 装配期已经喊过一次,逐次
调用再喊只会刷屏。判定结果不受此参数影响。
"""
# Phase 1: 调用方不表态 —— 与 False 严格区分,用模型默认档
if enable_thinking is None:
return {}
slot = profile.thinking_on if enable_thinking else profile.thinking_off
direction = "thinking_on" if enable_thinking else "thinking_off"
# Phase 2: 形态未知 —— 提供了开关却不知道怎么发,静默放行就是欺骗调用方
if slot is None:
raise ThinkingUnsupportedError(
f"provider {profile.name!r}{direction} 形态未知(模型 {model!r}): "
f"本库不知道该 provider 如何表达这一档。请用 register_provider 注册形态,"
f"或改用 SourceConfig.extra_body 直接下发供应商参数"
)
# Phase 3: 能力未登记 —— 新模型上线不该被库挡住,但也不该假装成功
if capability is None:
if warn_unregistered:
_warn_unregistered(model, profile, slot)
return slot
# Phase 4: 明确不支持关闭 —— 调用方要的是"不推理"的语义保证,给不了必须说
if enable_thinking is False and not capability.can_disable:
raise ThinkingUnsupportedError(
f"模型 {model!r} 无法关闭推理,enable_thinking=False 无法满足: "
f"{capability.evidence}。该模型的推理是固有属性,任何参数都关不掉——"
f"需要关闭思维链请换用支持关闭的模型"
)
return slot
def _warn_unregistered(model: str, profile: ProviderProfile, slot: Mapping[str, Any]) -> None:
logger.warning(
"模型 {} 的推理能力未登记,按 provider {} 的形态尽力注入 {};"
"若该模型实际不支持这一档,本次设置将静默失效。实测后请用 register_capability 登记",
model,
profile.name,
dict(slot),
)
def get_provider(
name: str, *, registry: Mapping[str, ProviderProfile] | None = None
) -> ProviderProfile:
+4 -1
View File
@@ -42,7 +42,8 @@ CREATE TABLE IF NOT EXISTS llm_calls (
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
cached_prompt_tokens INTEGER,
model_reported TEXT,
sampling TEXT
sampling TEXT,
reasoning_tokens INTEGER
);
"""
@@ -51,6 +52,7 @@ _BACKFILL = (
("cached_prompt_tokens", "ALTER TABLE llm_calls ADD COLUMN cached_prompt_tokens INTEGER"),
("model_reported", "ALTER TABLE llm_calls ADD COLUMN model_reported TEXT"),
("sampling", "ALTER TABLE llm_calls ADD COLUMN sampling TEXT"),
("reasoning_tokens", "ALTER TABLE llm_calls ADD COLUMN reasoning_tokens INTEGER"),
)
# 探测现有列;尊重 search_path(to_regclass 按当前 search_path 解析)
@@ -81,6 +83,7 @@ _COLUMNS = (
"cached_prompt_tokens",
"model_reported",
"sampling",
"reasoning_tokens",
)
_INSERT = (
+4 -1
View File
@@ -37,7 +37,8 @@ CREATE TABLE IF NOT EXISTS llm_calls (
created_at TEXT NOT NULL DEFAULT (datetime('now')),
cached_prompt_tokens INTEGER,
model_reported TEXT,
sampling TEXT
sampling TEXT,
reasoning_tokens INTEGER
);
"""
@@ -47,6 +48,7 @@ _BACKFILL_COLUMNS = (
("cached_prompt_tokens", "INTEGER"),
("model_reported", "TEXT"),
("sampling", "TEXT"),
("reasoning_tokens", "INTEGER"),
)
_COLUMNS = (
@@ -71,6 +73,7 @@ _COLUMNS = (
"cached_prompt_tokens",
"model_reported",
"sampling",
"reasoning_tokens",
)
_INSERT = (
+60 -8
View File
@@ -21,7 +21,14 @@ from polygateway.errors import (
SourceDeadError,
TransientError,
)
from polygateway.providers import ProviderProfile, get_provider
from polygateway.providers import (
ProviderProfile,
ThinkingCapability,
ThinkingUnsupportedError,
get_capability,
get_provider,
resolve_thinking,
)
from polygateway.streaming import StreamLivenessTimeout, stream_with_liveness_timeouts
from polygateway.types import EmbeddingTransportResult, SourceConfig, TransportResult
@@ -177,6 +184,25 @@ def _coerce_cached_tokens(usage: Any) -> int | None:
return cached
def _coerce_reasoning_tokens(usage: Any) -> int | None:
"""取 usage.completion_tokens_details.reasoning_tokens(issue #6);形态异常一律 None。
与 `_coerce_cached_tokens` 逐条同构(两者是 OpenAI 兼容 usage 里对称的一对):
`0` 如实保留、负数与非整数归 None、`bool` 显式排除。差别只在语义——本字段
的 None 是"**本次调用**未上报"而非"该源不上报": 中转在上游不返回 usage 时
会本地补算并整体替换 usage 对象,把 details 一并吃掉(findings §4c)。
"""
if not isinstance(usage, dict):
return None
details = usage.get("completion_tokens_details")
if not isinstance(details, dict):
return None
reasoning = details.get("reasoning_tokens")
if isinstance(reasoning, bool) or not isinstance(reasoning, int) or reasoning < 0:
return None
return reasoning
def _coerce_model_reported(value: Any) -> str | None:
"""取响应体的 model 字段(issue #3);非 str 或空白串一律 None,收口时去空白。
@@ -265,9 +291,14 @@ class OpenAICompatTransport:
self,
*,
registry: Mapping[str, ProviderProfile] | None = None,
capabilities: Mapping[str, ThinkingCapability] | None = None,
client_factory: Callable[[SourceConfig], httpx.AsyncClient] | None = None,
) -> None:
self._registry = registry
self._capabilities = capabilities
# 未登记模型只喊一次: 装配期已喊过,逐次调用再喊是日志洪水。
# 实例级而非模块级 —— 模块级可变状态违反纯 asyncio 中立铁律
self._warned_models: set[str] = set()
self._client_factory = client_factory or _default_client_factory
self._clients: dict[str, httpx.AsyncClient] = {}
@@ -290,10 +321,20 @@ class OpenAICompatTransport:
payload: dict[str, Any] = {"model": source.model, "messages": messages, "stream": stream}
if stream:
payload["stream_options"] = {"include_usage": True} # 强制 usage 帧(三项目同款)
if source.enable_thinking is True:
payload.update(profile.thinking_on)
elif source.enable_thinking is False:
payload.update(profile.thinking_off)
# 形态(provider 级)与能力(model 级)在此相遇;不可满足时 ValueError,
# 由 complete() 翻译为四分类之一(issue #5)
capability = get_capability(source.model, table=self._capabilities)
first_time = source.model not in self._warned_models
self._warned_models.add(source.model)
payload.update(
resolve_thinking(
profile,
capability,
source.enable_thinking,
model=source.model,
warn_unregistered=first_time,
)
)
# 顺序即优先级(issue #4 设计决策 A): 配置级 extra_body 在前,调用级
# overlay(含结构化注入)在后覆盖之。两行不可调换
payload.update(source.extra_body)
@@ -311,9 +352,18 @@ class OpenAICompatTransport:
) -> TransportResult:
"""一次原始调用;HTTP/线路/流式异常按 ARCH §6.2 翻译为领域错误。"""
profile = get_provider(source.provider, registry=self._registry)
payload = self._build_payload(
messages=messages, source=source, profile=profile, stream=stream, overlay=overlay
)
try:
payload = self._build_payload(
messages=messages, source=source, profile=profile, stream=stream, overlay=overlay
)
except ThinkingUnsupportedError as exc:
# 推理开关不可满足是**请求本身**的问题: 换源重试都救不了它。只捕这个
# 专用类型而非宽 catch ValueError —— 后者会把序列化等无关错误误贴标签
raise RequestRejectedError(
f"{source.name} 推理开关无法满足: {exc}",
source_name=source.name,
operation="chat",
) from exc
url = source.base_url.rstrip("/") + "/chat/completions"
client = self._client_for(source)
ctx: dict[str, Any] = {"source_name": source.name, "operation": "chat"}
@@ -400,6 +450,7 @@ class OpenAICompatTransport:
raw={"usage": sink.get("usage")},
cached_prompt_tokens=_coerce_cached_tokens(sink.get("usage")),
model_reported=_coerce_model_reported(sink.get("model")),
reasoning_tokens=_coerce_reasoning_tokens(sink.get("usage")),
)
def _check_done(
@@ -484,6 +535,7 @@ class OpenAICompatTransport:
raw={"usage": body.get("usage")},
cached_prompt_tokens=_coerce_cached_tokens(body.get("usage")),
model_reported=_coerce_model_reported(body.get("model")),
reasoning_tokens=_coerce_reasoning_tokens(body.get("usage")),
)
async def aclose(self) -> None:
+11 -1
View File
@@ -99,6 +99,15 @@ class LLMResponse:
model_reported: str | None = None
"""API 响应体里的 model 字段;None = 未上报。与 `model`(配置别名)可能
分叉——供应商把别名指向新权重时,实验复现必须认这个串。"""
reasoning_tokens: int | None = None
"""推理消耗的输出 token 数(含在 `completion_tokens` 内,故不影响成本总额,
只补归因;issue #6)。
`None` = **本次调用**未上报,**不是**"该源不上报"——中转网关在上游不返回
usage 时会用本地 tokenizer 补算并整体替换 usage 对象,把
`completion_tokens_details` 一并吃掉(findings §4c 实测同一请求 10 轮呈
6:4 双峰)。实测三家供应商在未推理时都是整个 details 缺失、无人上报 `0`,
故下游判据须为 `in (None, 0)`,写 `== 0` 的条件永远不成立。"""
@dataclass(frozen=True)
@@ -151,9 +160,10 @@ class TransportResult:
ttft_ms: float | None
max_inter_token_ms: float | None
raw: dict[str, Any]
# —— 可观测字段(issue #3;带默认值,非 OpenAI 兼容的 transport 可不填)——
# —— 可观测字段(issue #3/#6;带默认值,非 OpenAI 兼容的 transport 可不填)——
cached_prompt_tokens: int | None = None
model_reported: str | None = None
reasoning_tokens: int | None = None
@dataclass(frozen=True)