feat: collect reasoning_tokens from the provider usage payload (issue #6)
Reasoning tokens are already counted inside completion_tokens, so the cost total was never wrong -- what was missing is the attribution: how much of a call was spent thinking rather than answering. LLMResponse and TransportResult each gain a trailing reasoning_tokens field, and the telemetry port grows from 21 to 22 columns with the new column appended in both backends so fresh and migrated schemas keep the same physical order. None means this particular call did not report the field, not that the source never reports it: a relay that falls back to a local tokenizer replaces the whole usage object and drops completion_tokens_details. Downstream checks must therefore read "in (None, 0)"; no provider was observed reporting a literal zero.
This commit is contained in:
@@ -394,6 +394,81 @@ class TestObservabilityFields:
|
||||
assert set(result.raw) == {"usage"}
|
||||
|
||||
|
||||
class TestReasoningTokens:
|
||||
"""issue #6: 推理消耗的输出 token,与 issue #3 的 cached_tokens 对称。
|
||||
|
||||
实测三家供应商在"未推理"时是整个 completion_tokens_details 缺失,无人上报
|
||||
0;且中转在上游不返回 usage 时会本地补算并吃掉该对象。故 None 的语义是
|
||||
"本次调用未上报",不是"该源不上报"(findings §4c)。
|
||||
"""
|
||||
|
||||
def _reasoning_usage(self, reasoning):
|
||||
return {**_USAGE, "completion_tokens_details": {"reasoning_tokens": reasoning}}
|
||||
|
||||
async def test_stream_reads_reasoning_tokens(self):
|
||||
def handler(request):
|
||||
return _sse_stream(_chunk(content="ok"), _chunk(usage=self._reasoning_usage(7)))
|
||||
|
||||
result = await _complete(_transport_for(handler), _source())
|
||||
assert result.reasoning_tokens == 7
|
||||
|
||||
async def test_non_stream_reads_reasoning_tokens(self):
|
||||
def handler(request):
|
||||
return httpx.Response(
|
||||
200,
|
||||
json={
|
||||
"choices": [{"message": {"content": "42"}}],
|
||||
"usage": self._reasoning_usage(7),
|
||||
},
|
||||
)
|
||||
|
||||
result = await _complete(_transport_for(handler), _source(), stream=False)
|
||||
assert result.reasoning_tokens == 7
|
||||
|
||||
async def test_zero_reasoning_tokens_is_a_real_zero(self):
|
||||
"""0(上报了且确实没推理)与 None(本次未上报)必须可区分。"""
|
||||
|
||||
def handler(request):
|
||||
return _sse_stream(_chunk(content="ok"), _chunk(usage=self._reasoning_usage(0)))
|
||||
|
||||
result = await _complete(_transport_for(handler), _source())
|
||||
assert result.reasoning_tokens == 0
|
||||
|
||||
async def test_usage_without_details_is_none(self):
|
||||
def handler(request):
|
||||
return _sse_stream(_chunk(content="ok"), _chunk(usage=_USAGE))
|
||||
|
||||
result = await _complete(_transport_for(handler), _source())
|
||||
assert result.reasoning_tokens is None
|
||||
|
||||
@pytest.mark.parametrize("bad", ["abc", -1, True, 1.5, None, [], {"x": 1}])
|
||||
async def test_malformed_reasoning_tokens_degrade_to_none(self, bad):
|
||||
"""`True` 必须排除: Python 里 isinstance(True, int) 为真。"""
|
||||
|
||||
def handler(request):
|
||||
return _sse_stream(_chunk(content="ok"), _chunk(usage=self._reasoning_usage(bad)))
|
||||
|
||||
result = await _complete(_transport_for(handler), _source())
|
||||
assert result.reasoning_tokens is None
|
||||
|
||||
async def test_details_not_a_dict_is_none(self):
|
||||
def handler(request):
|
||||
usage = {**_USAGE, "completion_tokens_details": "oops"}
|
||||
return _sse_stream(_chunk(content="ok"), _chunk(usage=usage))
|
||||
|
||||
result = await _complete(_transport_for(handler), _source())
|
||||
assert result.reasoning_tokens is None
|
||||
|
||||
async def test_salvage_path_records_none_not_zero(self):
|
||||
"""打捞路径拿不到 usage 帧: 记 None(未知)而非 0(确定没推理)。"""
|
||||
|
||||
def handler(request):
|
||||
return _sse_stream(_chunk(content="ok"), done=False)
|
||||
|
||||
result = await _complete(_transport_for(handler), _source(missing_done="salvage"))
|
||||
assert result.reasoning_tokens is None
|
||||
|
||||
|
||||
class TestNonStreamFastPath:
|
||||
async def test_non_stream_parses_message(self):
|
||||
def handler(request):
|
||||
|
||||
Reference in New Issue
Block a user