From 0c1165965fb06ebf8a4b55c9e53dbdb68751e506 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 07:27:19 -0400 Subject: [PATCH 01/10] docs: record verified 1.3.4 publication and release checks --- CLAUDE.md | 2 ++ ...09-09-134-thinking-contracts-validation.md | 36 ++++++++++++++++++- 2 files changed, 37 insertions(+), 1 deletion(-) diff --git a/CLAUDE.md b/CLAUDE.md index dbf5fa0..a2870f8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -103,6 +103,8 @@ make ci # 只读验证(check + test) Gitea 包 registry 是 **owner 级**(`/iomgaa/-/packages/`)不是仓库级;PyPI 元数据不含仓库字段,故不会自动挂到 `PolyGateway/packages`,需在包页面手动 Link to a repository。 +发布验证补充(2026-09-09 实测):私有索引不镜像 setuptools,下载 sdist 时 pip 即使带 `--no-deps` 仍可能尝试安装构建依赖;已具备构建工具的 conda 环境可加 `--no-build-isolation`,仍须核对下载来源与散列。包关联 POST 本次返回400,但匿名包页面和认证 GET 包 API 均证实已关联目标仓库;**400本身不算成功,也不能直接推断已关联**,必须读取实际 `repository.full_name` 与页面链接。包 API 可能要求认证(匿名401),页面仍应匿名亲查;不要为重试关联先解绑现有正确关系。 + ### 4.5 配置管理 - 工程配置走 `pydantic-settings` + `.env`(模板 `.env.example`,敏感项不提交);严禁硬编码默认值;缺失关键配置直接报错。 - 多源命名约定 `{SCOPE}__{PROVIDER}__{N}__{FIELD}`;韧性参数键名沿用三项目习惯(`LLM_TIMEOUT` 等),降低迁移成本。 diff --git a/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md b/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md index e8ddaae..867fbd2 100644 --- a/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md +++ b/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md @@ -7,7 +7,7 @@ date: 2026-09-09 # 1.3.4 推理契约验证与发布准备 -> 最新状态(2026-09-09):发布准备完成;用户已正式批准将未补全模型矩阵、失败/UNKNOWN/不可达/缺轮及缺下游现行配置证据作为本版验收例外,保留原始结论而非 PASS。可移交合并与包发布;本轮未 merge/push/tag/构建/上传,合并后门与外部产物验收仍待执行。以下各节是分阶段历史,不追改当时结论;最新证据与例外见文末。原始输出在 `tests/outputs/134/`,不提交。 +> 最新状态(2026-09-09):**1.3.4 已发布并完成外部验收,#21/#25/#26 已评论关闭**。main/tag 指向 `af57f93adce24b43fd10b6d8e1281ab8ee43c0a8`;合并后门、下载独立安装及页面结果见文末。用户批准的未补全矩阵、FAIL/UNKNOWN/不可达/缺轮及缺下游配置例外保持原结论,不冒充 PASS。以下为分阶段历史,不追改当时结论;原始输出在 `tests/outputs/134/`,不提交。 ## 基线与修改边界 @@ -233,3 +233,37 @@ pi-lens 仍报非 conda 解释器缺 httpx/pytest/dotenv/pydantic 及旧 S conda 启动器既有 RequestsDependencyWarning 保留,不宣称零告警。本轮 conda 内实际导入 dotenv/httpx/redis.asyncio 成功;工具的非conda LSP旧诊断不转成代码修改或忽略规则。 **可移交合并与包发布,不等于已发布。** 父会话按本版例外边界完成本次文档/版本差异审查,再执行合并后静态/日常门及未豁免的发布检查;不把本次例外解释为必须补全模型矩阵,也不把豁免项勾成已跑通过。merge/push/tag/构建/twine上传/下载解包独立安装/Release及registry页面检查均尚未执行;只能在实际完成后记录。不得覆盖已有同版本不同字节。 + +## 1.3.4 发布完成与外部验收(2026-09-09) + +父会话已审 `dae12f9` 发布准备 diff,用户授权发布及指定例外;本轮无新生产/测试改动、不派子代理。已有红绿/变异与独立报告按适用范围复用,716份历史证据 SHA-256 复核不变。以下日志均相对 `tests/outputs/134/publish-20260909/`,命令无掩盖退出码管道;长跑由 tmux `pgw134-publish` 串行执行,`PYTHONUNBUFFERED=1`,每门独立 `.exit`。 + +| 步骤 | 实际结果与证据 | +| --- | --- | +| 远端核验/合并 | fetch后 origin/main=`a716f12` 无新变更;精确文件名检查 registry 无1.3.4,tag无占用、Release404。`--no-ff` 合并为 `af57f93adce24b43fd10b6d8e1281ab8ee43c0a8`,与批准 `dae12f9` 树零差异;`merge.log`/`merge-summary.log` | +| 合并后 `make lint` | exit0,ruff/import-linter 1 kept、0 broken,无自动修改;`lint.log/.exit` | +| 合并后 `make test` | **1518 passed、23 skipped、108 deselected,95%覆盖率,276.42秒,exit0**;`daily.log/.exit`。skip与deselected不计通过 | +| M2 AUTO定向 slow | **2 passed、80 deselected,46.85秒,exit0**;各5轮、并发1,空wire流式开启命题通过;`m2-auto.log/.exit`。M2.5 run `d4b158162ad543e5a4745d21b29ab03e`,M2.7 run `07a388cafa9f4fcf95dc0c6876fd0751`,均 planned=completed=5、无缺轮;实际逐轮/命题报告已读取 | +| Redis时间语义 slow | contracts/integration选择 **18 passed、156 deselected,1109.45秒,exit0**;`redis-time.log/.exit`。`gates.exit=0`,未重启全型号矩阵 | +| push与tag | main推送成功;注释tag对象 `0bdb0f70e4667a2fe50a567f0809a5a421c5442a`,解引用同上述main;`push-main.log`/`push-tag.log`/`remote-refs-after.log` | +| 构建/上传 | 核实项目绝对dist路径且不是链接后清旧1.3.3产物;conda `python -m build`、`twine check dist/*`、twine上传均exit0,wheel/sdist各一份。凭据只从tea内存读取进入TWINE_PASSWORD,无token argv/日志;`build`/`twine-check`/`upload` 的 `.log/.exit` | +| 独立下载/解包 | registry下载wheel与sdist,与本地构建SHA-256逐字匹配;包内五个关键生产文件与发布提交相同,wheel METADATA版本/Markdown正文、sdist README均与已发布README相符;`download-final.log/.exit`、`download-verification.json` | +| 独立安装与行为 | `/tmp/pgw134-registry-eyges3d7/installed/polygateway/__init__.py`,仓库外 `python -I` 并显式target来源验证;M2空AUTO、M3 AUTO在error/nearest均拒绝、M3 medium、同值/新增raw冲突与raw-only保留全部通过 | +| Release/包页面 | Release POST201,正文逐字来自CHANGELOG本节。匿名实际读取包页面、单版Release及Releases列表,均HTTP200、正文含迁移/例外;包页面有目标仓库链接,认证包API的 repository.full_name=`iomgaa/PolyGateway`;`pages-final.log/.exit`、`external-verification.json`及HTML/可见文本原件 | +| issue收尾 | 仅#21/#25/#26评论后关闭,GET各自核实closed;评论明确验证范围、失败例外、缓存迁移及“不识别所有渠道故障”;`issues.log/.exit`和`issue*-after.json`。未改#19/#22/#23/#24 | + +产物 SHA-256:wheel `86ad5cc025bfda127c4f0e4bbe1252d1bb8943b7678f9ff5561200bac71fe2a2`;sdist `85620f993ff9d19845088a071d251049e3dfae8d035c05652bdb62b4340a1a66`。 + +外部亲查地址:[包页面](https://gitea.iomgaa.online/iomgaa/-/packages/pypi/polygateway/1.3.4)、[v1.3.4 Release](https://gitea.iomgaa.online/iomgaa/PolyGateway/releases/tag/v1.3.4)、[Releases列表](https://gitea.iomgaa.online/iomgaa/PolyGateway/releases)。结果补记在独立 `docs/1.3.4-release-evidence` 分支,不改变已发布main/tag或覆盖同版本产物。 + +### 真实失败与恢复(原件保留,不改写成功) + +| 检查失败 | 根因/处置与边界 | +| --- | --- | +| 最初索引grep误报占用 | `1.3.4` 被当正则匹配历史sha片段;改用完整字面 `polygateway-1.3.4`,并核对tag/Release均不存在。没有实际同版占用,没有覆盖 | +| 首次sdist下载exit1 | 私有索引无setuptools,pip即使 `--no-deps` 仍启构建隔离;`download.log/.exit`保留。复用已下载wheel,sdist加 `--no-build-isolation` 使用现有conda构建工具后exit0;不新增依赖、不重传包 | +| 包link POST400 | 返回 `invalid argument`,`release-create.log/.exit`保留exit1;不把400当成功/不猜根因。匿名页面已有仓库链接,认证GET包API证实本版正确关联,故无需解绑重挂;Release创建201不重复执行 | +| 首次页面验证exit1 | 三个匿名页面及Release正文均通过,但额外包API匿名GET401;`pages.log/.exit`保留。仅该API沿现有tea认证GET后200,最终所有页面/API核对exit0;不扩大为发布包匿名不可下载 | + +上述操作差异已补入CLAUDE发布清单。conda启动器既有RequestsDependencyWarning仍保留。用户验收例外完全沿前节:embedding503与claude命题仍FAIL、M3非流UNKNOWN及其余未覆盖/缺下游资料未变;本轮成功不外推全矩阵或所有渠道。1.3.4发布任务到此停止,1.3.5由父会话继续。 + From e71a623b045453c198b4348dfdae4372c22720ba Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 07:28:02 -0400 Subject: [PATCH 02/10] docs: remove trailing blank line from release evidence --- .../findings/2026-09-09-134-thinking-contracts-validation.md | 1 - 1 file changed, 1 deletion(-) diff --git a/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md b/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md index 867fbd2..81e89b5 100644 --- a/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md +++ b/research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md @@ -266,4 +266,3 @@ conda 启动器既有 RequestsDependencyWarning 保留,不宣称零告警。 | 首次页面验证exit1 | 三个匿名页面及Release正文均通过,但额外包API匿名GET401;`pages.log/.exit`保留。仅该API沿现有tea认证GET后200,最终所有页面/API核对exit0;不扩大为发布包匿名不可下载 | 上述操作差异已补入CLAUDE发布清单。conda启动器既有RequestsDependencyWarning仍保留。用户验收例外完全沿前节:embedding503与claude命题仍FAIL、M3非流UNKNOWN及其余未覆盖/缺下游资料未变;本轮成功不外推全矩阵或所有渠道。1.3.4发布任务到此停止,1.3.5由父会话继续。 - From a81cc911245fb59d28c784660332942816c15252 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 09:08:47 -0400 Subject: [PATCH 03/10] docs: propose logical call statistics and failure diagnostics --- ...026-09-09-135-call-observability-design.md | 202 ++++++++++++++++++ 1 file changed, 202 insertions(+) create mode 100644 research-wiki/designs/2026-09-09-135-call-observability-design.md diff --git a/research-wiki/designs/2026-09-09-135-call-observability-design.md b/research-wiki/designs/2026-09-09-135-call-observability-design.md new file mode 100644 index 0000000..b374257 --- /dev/null +++ b/research-wiki/designs/2026-09-09-135-call-observability-design.md @@ -0,0 +1,202 @@ +# 1.3.5:逻辑调用统计与结构化失败诊断 + +> 状态:独立审查及定向复审通过,待人类批准;禁止据此实施公共 API。 +> 日期:2026-09-09。范围:issue #19、#23。基线源码 HEAD e71a623。 +> 复用现有 Emitter / schema / 三条治理循环,不引入追踪平台、不做 deadline / hedging、不在库内复制上下游关联状态。 + +## 1. 问题与边界 + +治理单位是一次逻辑调用,但当前 `latency_ms` / `call_id` 只描述**单次尝试**:重试几次、等了多久、缓存有没有参与,SQL 答不出来。诊断侧,transport 抛出的领域错误已带 `status_code` / `operation` / `body_text`(errors.py:52),进 emitter 前却被 `str()` 压平成一列自由文本。终态失败可能没有选中源,但 **scope 始终已知**。 + +中转把供应商 529 改写成 503 后,库只能如实记 503,不能猜回 529;未知成本、未知状态一律 NULL。本版不改重试预算、429 语义、stall、推理能力表与取消结算算法,不新增核心依赖。 + +## 2. 方案 + +| 方案 | 收益 | 代价 | +| --- | --- | --- | +| A:只在 RetryMW 加两个计数、错误列细分 | 补丁最小 | 漏缓存命中、结构化重问、embedding 分批与三类失败终态,回答不了"整次调用" | +| B:每调用局部统计 + 领域异常下沉到单一出口,复用现有遥测行 | 统一边界、字段保真、不默认加成功行、不双计费用 | 增加内部上下文、响应字段与 recorder 字段;须补齐缺失的失败终态行 | +| C:新增独立逻辑调用表 / 通用事件端口 | 完整追踪与任意事件分析 | 新存储与运维面,超出当前需求 | + +推荐 **B**。显式传递局部对象,不用 client 共享可变计数、不用模块级 ContextVar;不统一三条治理循环,只统一计数与诊断出口。 + +## 3. 逻辑边界与公开统计 + +四种响应(`LLMResponse`、`EmbeddingResponse`、`OcrTextResult`、`OcrLayoutResult`)追加 `call_stats: CallStats | None = None`。新增 frozen `CallStats` 并由包根导出——四份平铺字段会立刻漂移。 + +| CallStats 字段 | 语义 | +| --- | --- | +| `logical_call_id: str` | 每次公开调用一个 UUID;重试、重问、分批共享;不占用既有 `parent_call_id` | +| `attempts: int` | 准入后实际调用 transport 端口的次数;含免预算 429 与端口本地拒绝;**不是 HTTP 请求条数** | +| `total_latency_ms: int` | 从输入校验通过到返回/异常传播前的单调时钟快照;含缓存、等待、重问、分批、内联记账与资源收尾 | + +输入校验异常发生在统计边界之外,保持原行为。第三方合成响应的 `None` 表示未知,不得默认伪造 0。 + +- **缓存命中**:`attempts=0`、新 logical ID、本次缓存路径耗时;不回放历史统计。缓存持久化排除 `call_stats`(`_serialize` 显式剔除),`_rehydrate` 显式覆盖为 `None`——`_RESPONSE_FIELDS` 过滤会放行历史 dict,不覆盖就会有 dict 冒充 `CallStats`。缓存 key 白名单不变。 +- **embed 空输入**(embedding.py:180-190):合法零尝试,返回真实统计(`attempts=0`),**不写任何遥测行**——与 cache_hit 不同,不要按"必录"推断它有台账行。 +- **OCR layout** 的 POST + ZIP GET 在同一 transport 调用内(monkey_ocr.py:262-296),计 1 次尝试。 +- **chat 重问 / embedding 多批**计入同一上下文,不重置计数(重问经 `call_next` 重入 RetryMW,已核对)。 + +`latency_ms` / `call_id` / `parent_call_id` 语义不变;"总耗时减最后一次尝试耗时"不等于纯等待(含其他本地工作)。不在响应里挂每次尝试的明细列表,避免公共响应无界增长。 + +### 3.1 失败与取消 + +本版**不向异常对象附加可变 `call_stats`**:第三方可能复用同一异常实例,first-write-wins 会把首次调用的统计误读成本次,覆盖写则串扰;复制任意异常又保证不了构造签名与自定义属性。异常类型与分类原样保留,失败侧的统计走 §6 的终态行。 + +- 取消:`CancelledError` 保持原类型与语义,不在其上加字段;取消路径只**尽力**写一条终态,不 shield、不开新后台任务。 +- 任意内部非领域异常(编程错)原样传播,本版**不承诺**为其提供任何统计或终态行,也不偷偷改分类。 + +## 4. 最小内部接缝 + +`CallStats` 与私有可变 `_CallContext` 都落 `types.py`:`types.py` 不反向依赖实现层,不产生循环,也不动 import-linter 分层(`ports : types : errors` 并列最内层)。私有上下文只持计数、单调时钟与必要去重状态,不做 I/O。 + +`ChatRequest` 追加内部上下文字段(`default=None, compare=False, repr=False`);`StructuredMW` 的 `dataclasses.replace` 保留同一引用(structured.py:96-110,已核对)。响应统计只在公共出口经 `replace` 附加。`GatewayClient` 需自存注入的 `now`(client.py 现未保存),**冻结快照是同步动作,不 await**。 + +- 上下文创建/冻结:`GatewayClient` 在公开入口创建、`finally` 冻结;`RetryMW` 只在 transport 调用前登记一次尝试。 +- **OCR 例外(M1)**:`image` 的类型/空校验在 `_call` 内(ocr.py:238-241)而非公开方法,故上下文在该校验**通过之后**创建,§3 的"校验在边界外"对 OCR 才成立。 +- `EmbeddingClient` / `OcrClient` 显式把同一上下文传到每批/每次尝试;统计生效与否**不由 telemetry 是否启用决定**。 +- 每次尝试 ID 仍在当前循环产生,与上下文的 logical ID 一起交给 Emitter。不持久化断点,不引入任务恢复。 + +## 5. 诊断字段与归因方式 + +保留现有 `error` 字符串供人阅读;Emitter 改为接受**领域异常对象**而非调用方先 `str()`,由单一 helper 提取有限诊断字段。普通超时/网络翻译在 transport 侧保留直接 `__cause__` 类型,空 `str()` 退回类名;**不遍历任意异常对象、不猜测正文**。 + +| 新增 INSERT 列 | 值域 / 来源 | +| --- | --- | +| `scope` | 配置池名,构造期注入 Emitter(见下);不拿 `source_name` 顶替 | +| `operation` | **公开方法固定四值**:`chat` / `embed` / `recognize_text` / `parse_layout`;由三个 client 在调用点给定 | +| `logical_call_id` | 当前调用上下文的 UUID;上下文缺席(库内现场构造的 `ChatRequest`)→ **NULL,不造 ID** | +| `event_kind` | `attempt` / `cache_hit` / `terminal_failure`;旧行 NULL,不回填 | +| `http_status_code` | **仅 attempt 行**:失败异常实收状态;无 HTTP 或未知 NULL;成功行不统一填 200 | +| `error_type` | 该行自身错误的领域类名;成功行 NULL | +| `cause_type` | **仅 attempt 行**:transport 直接捕获的底层异常类名,未知 NULL | +| `error_body` | **仅 attempt 行**:既有 `summarize_body` 有界摘要,未知 NULL;不存 `raw_text`、不存全量原文 | +| `attempts` / `total_latency_ms` | **仅 terminal_failure 行**填逻辑快照,其余行 NULL;快照在写入前冻结 | + +全部可空,追加到物理列末尾(与旧表 ALTER 追加位置一致,`schema.py` 的 DDL / BACKFILL / COLUMNS 三处同改)。 + +**归因方式(C1,决定性)**:`GatewayUnavailableError` 家族从不携带 `status_code` / `body_text`(errors.py:118-166),终态行的 `http_status_code` / `cause_type` / `error_body` 因此**保持 NULL,这是它自身的真实状态**——不把最后一次 attempt 的状态码与正文搬上来伪装成整池诊断(那正是"不拿最后一个源冒充整池归因"的同一条红线)。终态行的 `error_type` 落它自己的类名(`AllSourcesExhausted` / `CircuitOpenError` / …),scope 级 reason 沿用**已有 error 文案**(`str(exc)` 已是 `"{scope} 网关暂时不可用: {reason}"`,不新增列)。逐源现场由同一 `logical_call_id` 的 attempt 行给出。 + +```sql +-- #19 验收:一次逻辑失败调用的完整现场(终态 + 各次尝试) +SELECT event_kind, source_name, http_status_code, error_type, cause_type, error_body, error +FROM llm_calls WHERE logical_call_id = :lcid ORDER BY created_at; +``` + +该查询必须同时给出"整池为何失败"(终态行 error 文案)与"每个源怎么死的"(attempt 行状态码/正文),测试按它断言。因此**本版不再为 `reason` / `per_source_reasons` 扩列**。 + +**结构化耗尽的可定位性(C2)**:`ResultInvalidError("结构化输出阶梯耗尽")` 的 message 不含 `validation_errors` / `repair_error`(structured.py:80-86),而该失败发生在 StructuredMW 之上——RetryMW 侧的 attempt 行全是成功行,终态行是唯一记录。做法:**不写 `error_body`**(该列只属 attempt),由 Emitter 的单一 helper 对 `ResultInvalidError` 生成**有界结构化说明**并入现有 `error` 字符串,复用 structured.py 已有的取材口径(至多 3 条、每条 200 字符,与 `_format_errors` 同参数,拼装函数收敛在 helper 一处)。`raw_text` **不重复落库**(它是模型正文,attempt 行的 `response` 列已按 `text_cap` 记过一份;再存一份等于绕过既有正文预算)。该说明可能包含模型输出片段,故遵循与 `error` 现有正文相同的隐私边界,不额外扩大保留范围。 + +**operation 的数据源(I1/I2)**:openai_compat.py:169 的 `_status_to_error` 硬编码 `operation="chat"`,而 `embed()` 的非 200 分支(:512、:527)也走它 → 现存所有 embedding HTTP 失败的 `exc.operation` 都是错的。修正:`_status_to_error` 增 keyword `operation`,`embed` 传 **`"embedding"`**(沿用该异常侧既有词表,不改 `chat` / `ocr_text` / `parse` / `download_result` 四值)。新列 `operation` 与 `exc.operation` 是**两个语义**:前者是公开方法,后者是 HTTP 子操作;新列由调用点给定,**绝不读 `exc.operation`**,两者不做自动转换。OCR 两个公开方法各自在调用级给定自己的值。 + +**scope 注入(M6)**:`TelemetryEmitter` 现在不知道 scope,`CacheMW` / `TelemetryMW` 自己也拿不到。构造期注入(三个 client 各一行),使 attempt / cache_hit / terminal_failure 三类行都带 scope,避免改三条调用链。model / provider / source 未选出时仍留原空值。 + +**成功侧不加承诺**:成功行不承诺 HTTP 状态与错误体;**本版不宣称 SQL 可直接统计所有成功逻辑调用的总耗时**(成功不加终态行)。`error_body` 沿用 `summarize_body` 的既有上限,**不纳入 `PGW_TELEMETRY_TEXT_CAP` 覆盖面**(该键现覆盖四处,详见 §8)。 + +## 6. 行语义与终态:只补确实缺失的失败 + +保留每次 attempt 与 cache_hit 的既有行,**不为成功新增终态行**。终态行不得复制已有 attempt 的 token 与成本。所有统计边界内的领域失败均尝试写终态,**包括已有 attempt 错误行的直接 RequestRejectedError/ResultInvalidError**;不再沿用“该类错误已录所以外层不录”的旧假设,400 密集负载的错误行可能翻倍,调用失败计数必须只取 terminal_failure。 + +**不变量(I3,统一措辞)**: + +| 结束形态 | 终态行数 | +| --- | --- | +| 以**领域错误**结束的逻辑调用 | 每次调用**尝试写一条**;持久化 best effort(recorder 写失败按既有降级只落 warning),故 SQL 可见行数 ≤ 1 | +| 取消 | 三个 client **同策略尽力写一条**,允许 0 条 | +| 非领域异常(编程错) | **0 条**,原样传播,本版无统计保证 | + +需要补的路径:chat 结构化耗尽(发生在 transport 成功之后,现无任何失败行)、embedding / OCR 的无源、准入拒绝、重试耗尽与尝试外取消。**取消口径统一(I4)**:chat 现由 TelemetryMW 对任何取消补终态(telemetry.py:487-495,含尝试内取消),embedding / OCR 按同一口径尽力补,避免下游按 `event_kind` 统计取消时拿到路径相关的结果。 + +终态与 attempt **不是重复事实**(前者描述逻辑终态,后者描述尝试),用 `event_kind` 区分;**禁止按 `error IS NOT NULL` 跨两类直接计失败调用次数**。chat 现有 TelemetryMW 终态路径收敛到公开边界的单一 helper,避免两处同时写;embedding / OCR 复用该 helper。 + +**终态行的请求摘要(M4)**:复用既有 200 字符输入摘要口径,描述**本次调用的整体输入**,但不扩大单行正文预算——embedding 终态取 `` 计数占位 + 第一批(至多 `batch_size` 条、每条 200 字符,与逐批行同款构造);OCR 终态沿用 `` 占位,图像 bytes 永不入库。**失败批的具体文本由同 `logical_call_id` 的 attempt 行给出**,终态行不保存全量原输入。 + +**错误文本口径(I7)**:OCR 现落 `"类名: msg"`(ocr.py:445-449,按类名归组的既有 metric 口径),chat / embed 落裸 `str(exc)`。改成"Emitter 收异常对象"后,该差异由统一出口的**显式文本策略参数**保留(OCR 保留类名前缀),**不再在三处复制参数列表**。 + +**取消时的终态写(残余风险,显式定策)**:该 `await` 本身是新的取消点。策略是**取消优先、不屏蔽**:外部取消落在这一 await 上时,`CancelledError` 照常传播(调用方可能因此看到 `CancelledError` 而非领域错误,与 TelemetryMW 现有行为同款);不 shield、不建新后台任务。冻结统计快照是同步动作,不 await。 + +终态行成本 NULL、usage `unavailable`;聚合费用仍只由 attempt / cache_hit 行决定。终态快照在写入前冻结,故不含自身写入耗时;成功响应快照包含其返回前已完成的内联遥测耗时。失败异常不附快照,不为对齐再 UPDATE 旧行。 + +取消 attempt 既有字符串 `"cancelled"` 保留:Emitter接收 `PolyGatewayError | str | None`,字符串不解析猜测诊断,error_type/cause_type/http_status_code/error_body均NULL;终态取消同样使用明确取消文案。终态既有latency_ms与新增total_latency_ms取同一冻结快照,避免双时钟微差。 + +## 7. recorder 兼容:装配期机械闸(C3) + +`TelemetryEmitter._record` 的 `except Exception` 会把旧 recorder 的 `TypeError` 吞成 warning(telemetry.py:463),后果是自定义 recorder 在下游升级后**100% 丢遥测且调用照常成功**——正是"遥测必录"要防的形态。文档级迁移清单挡不住它。 + +机械闸:在 `TelemetryEmitter.__init__`(三个 client 的唯一汇流点,与 `text_cap` 值域校验同处)对 `recorder.record_llm_call` 做**一次** `inspect.signature(...).bind(**<完整新 kwargs 形状>)`,**不执行写入**;含 `**kwargs`(`VAR_KEYWORD`)者自动通过。校验失败 → 装配期抛错。若目标不可 inspect(C 实现等),同样按**配置错误**当场报错,不进入"运行期静默丢行"。 + +边界诚实声明:签名 bind 只证明该形状能被接受,**不能证明函数体真的落这些列**;这是一道装配闸,不是行为验证。它既不是被否掉的 `runtime_checkable` 判定,也不是捕 `TypeError` 重试写入。是否扩主 `TelemetryRecorder` Protocol(备选:独立扩展端口保留旧实现)是 §10 的人类批准项;本草案不同时实现两套接口,自带两个 recorder 的 `**fields` 签名可接受新增参数,但其 schema、INSERT 字段与契约测试仍须同步,不能称后端完全不受影响。 + +校验所用参数名从 `TelemetryRecorder.record_llm_call` 的协议签名派生,不手抄第四份字段清单;只用哨兵值做bind形状校验,不读取真实请求数据。结构化说明的限条数/限长复用现有常量,若需命名常量则在既有规则所有者中定义并由两个消费者引用,不复制数值。 + +## 8. 下游可见变更与文档同步 + +**SQL 迁移清单(I6,批准项须按此逐条看)**: + +| 影响面 | 变化 | 下游动作 | +| --- | --- | --- | +| 失败行数 | 新增 `terminal_failure` 行(每失败调用至多 1) | 计失败调用改 `WHERE event_kind = 'terminal_failure'` | +| `error IS NOT NULL` | 同时命中 attempt 与 terminal 两类 | 不再作为"失败调用数"的判据 | +| `AVG(latency_ms)` | terminal 行携带**逻辑总耗时**,量级大于单次尝试 | 时延看板一律按 `event_kind` 分组或过滤 | +| 费用聚合 | terminal 行 `cost` 恒 NULL、usage `unavailable` | 费用仍只由 attempt / cache_hit 行决定,口径不变 | +| 成功侧 | **不加**任何成功汇总行 | 成功逻辑调用总耗时仍从响应 `call_stats` 读,不从 SQL 读 | +| `http_status_code` | 失败行上可能是 200(monkey_ocr.py:265-270 的 `success != true` 带 200 上抛) | 该列不可作失败判据 | + +**文档同步(M5,发布前必须同批)**:README:23 的"必录 26 字段"、README:399 与 ARCHITECTURE.md:592 的 `PGW_TELEMETRY_TEXT_CAP` 覆盖面四处枚举(须明确:`error_body` 沿用 `summarize_body` 上限,`error` 保留既有文本口径并仅对新增结构化说明限长;二者**不在 cap 覆盖内**)、ARCHITECTURE.md:565 的必录字段清单与 §7.8 补列一节、`.env.example` 相关注释、CHANGELOG 与 wiki(docs-convention §2)。**字段数与物理列数一律以 `inspect.signature` / `len(COLUMNS)` 实测改写,不凭记忆**(现状:26 个 INSERT 字段 + `created_at` = 27 物理列;本版新增 10 列)。 + +## 9. 非功能与测试矩阵 + +| 维度 | 要求 | +| --- | --- | +| 并发 | 每调用独立对象;同一 client 并发不串 logical ID / 计数 / 统计;无全局状态 | +| 取消 | 各等待点穿透;`finally` 释放既有资源;终态写取消优先;不新增 shield 与后台任务 | +| 降级 | recorder 写失败不改统计与主结果;准入后端仍 fail-closed;收尾失败仍原 warning | +| 持久化 | schema 单一事实源;SQLite auto / PG manual 裁剪 INSERT 保持;不 ALTER 默认生产 PG、不改旧列、不回填旧行 | +| 幂等 | attempt ID 唯一,终态独立 ID,不重复写同一终态;缓存命中不复制历史统计;不 UPDATE 计费 | + +验收优先离线:真实 client + 内存后端 + FakeClock + MockTransport + 临时 SQLite,复用 1.3.4 设施;不重跑未变的模型能力矩阵。 + +| 测试族 | 必须证明 | +| --- | --- | +| logical 计数 | 一次成功、失败重试、免预算 429、多源拒绝、缓存命中、结构化重问、embedding 多批、OCR 双 HTTP、空输入(0 尝试且 0 遥测行) | +| 计时 | 缓存 IO、退避、准入等待、重问、收尾均计入;关闭 recorder 仍正确;毫秒/秒不混用 | +| 失败与取消 | 领域失败恰一次尝试写终态;取消三条路径同策略(允许 0 行);**终态写 await 上被取消 → `CancelledError` 传播**;permit / 探针释放不变;非领域异常 0 行且原样传播 | +| 保真诊断 | 503 包装不改回 529;直接 529 记 529;空 Connect/Read/Write/PoolTimeout 文案有类型;embedding HTTP 失败的 `exc.operation` 为 `embedding`;新列 `operation` 恒为四值之一且不随异常变化 | +| 归因 SQL | §5 那条按 `logical_call_id` 的查询同时给出终态 reason 文案与逐源状态码/正文;终态行三列为 NULL;结构化耗尽的 `error` 含有界 validation/repair 说明且不含 `raw_text` | +| 行语义 | 三类行均带 scope;每失败调用至多一条终态;attempt 与 terminal 区分;费用不重复;`AVG(latency_ms)` 按 `event_kind` 分组的断言 | +| 装配闸 | 旧签名 recorder → 装配期报错(非 warning);`**kwargs` recorder 通过;不可 inspect → 配置错误 | +| 存储兼容 | SQLite 新旧表、PG manual 缺列裁剪 / auto 追加、旧行 NULL、新旧进程混写 | +| 变异 | 计数位置、上下文复制、缓存历史回放、提前 `str()` 压平、终态双计费用分别红→绿;PG 真实集成 + 常规全套 + 独立验证 | + +## 10. 集中人类批准项 + +| # | 决策 | 推荐 | +| --- | --- | --- | +| 1 | 公开响应结构 | 一个 `CallStats` 对象而非四类响应各铺三字段;需确认命名与消费便利性 | +| 2 | 失败侧统计范围 | 不改/不复制异常;失败统计只落终态行,调用方仅在成功响应读 `call_stats`;需确认该取舍可接受 | +| 3 | recorder 兼容路线 | 扩主 `TelemetryRecorder` Protocol + 装配期 bind 闸;备选独立扩展端口保留旧实现;需确认是否存在必须兼容的自定义 recorder | +| 4 | 新增失败终态行 | 补漏但不加成功汇总;须批准 §8 表中**全部五项**下游可见变化(不止行数) | +| 5 | 取消时终态写取消优先 | 调用方可能看到 `CancelledError` 而非领域错误(同 TelemetryMW 现状);需确认接受 | +| 6 | `operation` 值域与异常侧修正 | 新列固定四值;`_status_to_error` 增 `operation` 参数、`embed` 传 `embedding`(修正现存误标,属下游可见的历史数据口径变化) | + +## 11. 独立审查处理表 + +| 项 | 结论 | 落点 | +| --- | --- | --- | +| C1 终态缺诊断来源 | **不采纳"定向读 `__cause__` / `per_source_reasons` 扩列"**;终态保留自身 NULL 状态,归因由 `logical_call_id` 关联 attempt 行 + 既有 reason 文案完成,并写死 SQL 验收 | §5 | +| C2 结构化耗尽可定位 | 采纳(变形):有界说明并入现有 `error` 字符串,不写 `error_body`、不重复存 `raw_text` | §5 | +| C3 recorder 静默失败 | 采纳:装配期一次 `signature.bind` 形状校验,含 `**kwargs`,不可 inspect 即配置错误;明确不验证函数体 | §7 | +| I1 / I2 operation 污染与未归一 | 采纳:新列固定四值由调用点给定,绝不读 `exc.operation`;只修 embedding 误标,其余异常侧词表不动 | §5 | +| I3 / I4 终态不变量与取消口径 | 采纳(定稿措辞):领域失败每调用尝试写一条、持久化 best effort;取消三路同策略尽力允许 0;非领域异常 0 条 | §6 | +| I5 上下文缺席语义 | 采纳:`None` → NULL,不造 ID | §5 | +| I6 迁移影响不止行数 | 采纳:列全五项 SQL 影响并进批准项 | §8、§10 | +| I7 OCR error 文本口径 | 采纳:保留类名前缀,由统一出口的显式文本策略参数承载,不复制参数列表 | §6 | +| M1 OCR 校验与统计边界 | 采纳:上下文在 `image` 校验通过后创建 | §4 | +| M2 embed 空输入 | 采纳:0 尝试且不写任何遥测行 | §3 | +| M3 失败行可能带 200 | 采纳:写明该列不可作失败判据 | §8 | +| M4 终态请求摘要未定 | 采纳:复用 200 字符口径,终态描述整体输入,不扩预算,不存全量原输入 | §6 | +| M5 文档同步缺项 | 采纳:列全六处并要求实测改数字 | §8 | +| M6 Emitter 不知 scope | 采纳:构造期注入 | §5 | +| 残余风险(终态写成新取消点) | 采纳:显式定策"取消优先不屏蔽",并进测试族与批准项 | §6、§9、§10 | + +自审:方案 B 复用既有 Emitter / schema 与三条循环,不需要 #22 / #24 的新调度。未采纳的两项(异常 first-write-wins、终态搬运最后一次 attempt 的状态与正文)理由已写在正文,不是遗漏。本文档无代码实现与测试通过声明;**§10 六项未获人类批准前不得实施公共 API**。 From 300ced5dbd6a35f8328444c2e3d9bdb6fb848052 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 09:42:24 -0400 Subject: [PATCH 04/10] docs: record approved call observability design and plan --- ...026-09-09-135-call-observability-design.md | 6 +- research-wiki/graph/edges.json | 7 + research-wiki/index.md | 10 +- research-wiki/log.md | 2 + .../2026-09-09-135-call-observability.md | 455 ++++++++++++++++++ 5 files changed, 473 insertions(+), 7 deletions(-) create mode 100644 research-wiki/plans/2026-09-09-135-call-observability.md diff --git a/research-wiki/designs/2026-09-09-135-call-observability-design.md b/research-wiki/designs/2026-09-09-135-call-observability-design.md index b374257..8f2c826 100644 --- a/research-wiki/designs/2026-09-09-135-call-observability-design.md +++ b/research-wiki/designs/2026-09-09-135-call-observability-design.md @@ -1,6 +1,6 @@ # 1.3.5:逻辑调用统计与结构化失败诊断 -> 状态:独立审查及定向复审通过,待人类批准;禁止据此实施公共 API。 +> 状态:独立审查及定向复审通过,**人类已于 2026-09-09 正式批准**(§10 六项批准项全数获批,可据此实施公共 API)。实施计划见 `research-wiki/plans/2026-09-09-135-call-observability.md`。 > 日期:2026-09-09。范围:issue #19、#23。基线源码 HEAD e71a623。 > 复用现有 Emitter / schema / 三条治理循环,不引入追踪平台、不做 deadline / hedging、不在库内复制上下游关联状态。 @@ -87,7 +87,7 @@ FROM llm_calls WHERE logical_call_id = :lcid ORDER BY created_at; **结构化耗尽的可定位性(C2)**:`ResultInvalidError("结构化输出阶梯耗尽")` 的 message 不含 `validation_errors` / `repair_error`(structured.py:80-86),而该失败发生在 StructuredMW 之上——RetryMW 侧的 attempt 行全是成功行,终态行是唯一记录。做法:**不写 `error_body`**(该列只属 attempt),由 Emitter 的单一 helper 对 `ResultInvalidError` 生成**有界结构化说明**并入现有 `error` 字符串,复用 structured.py 已有的取材口径(至多 3 条、每条 200 字符,与 `_format_errors` 同参数,拼装函数收敛在 helper 一处)。`raw_text` **不重复落库**(它是模型正文,attempt 行的 `response` 列已按 `text_cap` 记过一份;再存一份等于绕过既有正文预算)。该说明可能包含模型输出片段,故遵循与 `error` 现有正文相同的隐私边界,不额外扩大保留范围。 -**operation 的数据源(I1/I2)**:openai_compat.py:169 的 `_status_to_error` 硬编码 `operation="chat"`,而 `embed()` 的非 200 分支(:512、:527)也走它 → 现存所有 embedding HTTP 失败的 `exc.operation` 都是错的。修正:`_status_to_error` 增 keyword `operation`,`embed` 传 **`"embedding"`**(沿用该异常侧既有词表,不改 `chat` / `ocr_text` / `parse` / `download_result` 四值)。新列 `operation` 与 `exc.operation` 是**两个语义**:前者是公开方法,后者是 HTTP 子操作;新列由调用点给定,**绝不读 `exc.operation`**,两者不做自动转换。OCR 两个公开方法各自在调用级给定自己的值。 +**operation 的数据源(I1/I2)**:openai_compat.py:169 的 `_status_to_error` 硬编码 `operation="chat"`,而 `embed()` 的非 200 分支(:512)也走它 → 现存所有 embedding HTTP 失败的 `exc.operation` 都是错的。(行号勘误 2026-09-09:本句原写 "(:512、:527)",实测 `:527` 属 `_complete_stream`即流式 chat,`embed` 只有 `:512` 一处;归属以实施计划 §2 表为准,决策未变。)修正:`_status_to_error` 增 keyword `operation`,`embed` 传 **`"embedding"`**(沿用该异常侧既有词表,不改 `chat` / `ocr_text` / `parse` / `download_result` 四值)。新列 `operation` 与 `exc.operation` 是**两个语义**:前者是公开方法,后者是 HTTP 子操作;新列由调用点给定,**绝不读 `exc.operation`**,两者不做自动转换。OCR 两个公开方法各自在调用级给定自己的值。 **scope 注入(M6)**:`TelemetryEmitter` 现在不知道 scope,`CacheMW` / `TelemetryMW` 自己也拿不到。构造期注入(三个 client 各一行),使 attempt / cache_hit / terminal_failure 三类行都带 scope,避免改三条调用链。model / provider / source 未选出时仍留原空值。 @@ -199,4 +199,4 @@ FROM llm_calls WHERE logical_call_id = :lcid ORDER BY created_at; | M6 Emitter 不知 scope | 采纳:构造期注入 | §5 | | 残余风险(终态写成新取消点) | 采纳:显式定策"取消优先不屏蔽",并进测试族与批准项 | §6、§9、§10 | -自审:方案 B 复用既有 Emitter / schema 与三条循环,不需要 #22 / #24 的新调度。未采纳的两项(异常 first-write-wins、终态搬运最后一次 attempt 的状态与正文)理由已写在正文,不是遗漏。本文档无代码实现与测试通过声明;**§10 六项未获人类批准前不得实施公共 API**。 +自审:方案 B 复用既有 Emitter / schema 与三条循环,不需要 #22 / #24 的新调度。未采纳的两项(异常 first-write-wins、终态搬运最后一次 attempt 的状态与正文)理由已写在正文,不是遗漏。本文档无代码实现与测试通过声明;**§10 六项已于 2026-09-09 获人类批准**,实施边界与红绿证据要求以上述实施计划为准。 diff --git a/research-wiki/graph/edges.json b/research-wiki/graph/edges.json index 485d438..1c6290f 100644 --- a/research-wiki/graph/edges.json +++ b/research-wiki/graph/edges.json @@ -465,6 +465,13 @@ "relation": "implements", "evidence": "复用既有26字段,四种行来源与无推理成败NULL;不增加生产数据面", "added": "2026-09-09T06:32:06.527808+00:00" + }, + { + "source": "plan:2026-09-09-135-call-observability", + "target": "design:2026-09-09-135-call-observability-design", + "relation": "implements", + "evidence": "1.3.5 逻辑调用统计与结构化失败诊断实施计划落地已批准设计", + "added": "2026-09-09T13:42:00.298915+00:00" } ] } \ No newline at end of file diff --git a/research-wiki/index.md b/research-wiki/index.md index 6d27ba9..471d247 100644 --- a/research-wiki/index.md +++ b/research-wiki/index.md @@ -1,8 +1,8 @@ # Research Wiki 索引 -> 自动生成,更新时间:2026-09-09 06:32 UTC +> 自动生成,更新时间:2026-09-09 13:42 UTC -## design (42) +## design (43) - [1.3.4 推理意图与测试证据设计](designs/2026-09-09-134-thinking-contracts-design.md) `design:2026-09-09-134-thinking-contracts-design` - [2026-07-20-m1-core-design](designs/2026-07-20-m1-core-design.md) `design:2026-07-20-m1-core-design` @@ -24,6 +24,7 @@ - [2026-08-19-issue14-admission-wait-policy-design](designs/2026-08-19-issue14-admission-wait-policy-design.md) `design:2026-08-19-issue14-admission-wait-policy-design` - [2026-08-24-issue15-telemetry-pool-lifecycle-design](designs/2026-08-24-issue15-telemetry-pool-lifecycle-design.md) `design:2026-08-24-issue15-telemetry-pool-lifecycle-design` - [2026-09-04-reasoning-effort-design](designs/2026-09-04-reasoning-effort-design.md) `design:2026-09-04-reasoning-effort-design` +- [2026-09-09-135-call-observability-design](designs/2026-09-09-135-call-observability-design.md) `design:2026-09-09-135-call-observability-design` - [est_tokens 解耦: 拆分限流预扣与遥测用量兜底(issue #2)](designs/est-tokens-decoupling.md) `design:est-tokens-decoupling` - [GatewaySettings 装配校验补齐(第二轮)](designs/settings-invariants-round-2.md) `design:settings-invariants-round-2` - [GatewaySettings 跨字段不变量守卫的生效范围](designs/settings-invariant-guards.md) `design:settings-invariant-guards` @@ -49,7 +50,7 @@ ## finding (15) -- [1.3.4 T0–T4 与 T7 确定性验证](findings/2026-09-09-134-thinking-contracts-validation.md) `finding:2026-09-09-134-thinking-contracts-validation` +- [1.3.4 推理契约验证与发布准备](findings/2026-09-09-134-thinking-contracts-validation.md) `finding:2026-09-09-134-thinking-contracts-validation` - [2026-07-20-m2-soak-workload](findings/2026-07-20-m2-soak-workload.md) `finding:2026-07-20-m2-soak-workload` - [2026-07-21-m25-acceptance](findings/2026-07-21-m25-acceptance.md) `finding:2026-07-21-m25-acceptance` - [2026-07-21-p6-soak-baseline](findings/2026-07-21-p6-soak-baseline.md) `finding:2026-07-21-p6-soak-baseline` @@ -65,9 +66,10 @@ - [P7 OCR soak 验收: 99.73% 与 13 不变量全 PASS](findings/p7-ocr-soak.md) `finding:p7-ocr-soak` - [推理开关与 reasoning_tokens: 供应商实测与业界做法](findings/2026-08-02-thinking-switch-and-reasoning-tokens.md) `finding:2026-08-02-thinking-switch-and-reasoning-tokens` -## plan (37) +## plan (38) - [1.3.4 推理契约实施计划](plans/2026-09-09-134-thinking-contracts.md) `plan:2026-09-09-134-thinking-contracts` +- [1.3.5 逻辑调用统计与结构化失败诊断实施计划](plans/2026-09-09-135-call-observability.md) `plan:2026-09-09-135-call-observability` - [2026-07-20-m1-core-plan](plans/2026-07-20-m1-core-plan.md) `plan:2026-07-20-m1-core-plan` - [2026-07-20-m2-distributed-plan](plans/2026-07-20-m2-distributed-plan.md) `plan:2026-07-20-m2-distributed-plan` - [2026-07-21-m25-resilience-plan](plans/2026-07-21-m25-resilience-plan.md) `plan:2026-07-21-m25-resilience-plan` diff --git a/research-wiki/log.md b/research-wiki/log.md index 1f4a0b1..d0fa0ca 100644 --- a/research-wiki/log.md +++ b/research-wiki/log.md @@ -156,3 +156,5 @@ - [2026-09-09 05:39 UTC] 重建索引: 98 篇页面 - [2026-09-09 06:32 UTC] 新增边: schema:llm-calls --implements--> design:2026-09-09-134-thinking-contracts-design - [2026-09-09 06:32 UTC] 重建索引: 98 篇页面 +- [2026-09-09 13:42 UTC] 新增边: plan:2026-09-09-135-call-observability --implements--> design:2026-09-09-135-call-observability-design +- [2026-09-09 13:42 UTC] 重建索引: 100 篇页面 diff --git a/research-wiki/plans/2026-09-09-135-call-observability.md b/research-wiki/plans/2026-09-09-135-call-observability.md new file mode 100644 index 0000000..7acc7a7 --- /dev/null +++ b/research-wiki/plans/2026-09-09-135-call-observability.md @@ -0,0 +1,455 @@ +--- +type: plan +node_id: plan:2026-09-09-135-call-observability +title: "1.3.5 逻辑调用统计与结构化失败诊断实施计划" +date: 2026-09-09 +--- + +# 1.3.5 逻辑调用统计与结构化失败诊断实施计划 + +> 设计:`research-wiki/designs/2026-09-09-135-call-observability-design.md`,**人类于 2026-09-09 正式批准**(§10 六项批准项全数获批)。 +> 计划审核门:Claude 自审 + **独立模型替代审查**(用户已同意以独立模型替代 Codex 那一道,结论仍须逐条核验后就地修订);plan 无人类门,审毕直接执行。 +> 目标:把治理单位从"一次尝试"补齐到"一次逻辑调用"——四种响应带 `CallStats`,遥测补 10 列诊断/归因字段与 `terminal_failure` 行(issue #19、#23)。 +> 方案:设计方案 B——每调用一个显式传递的私有可变上下文 + 领域异常下沉到单一遥测出口;复用既有 Emitter / schema / 三条治理循环,不新增表、不新增依赖、不引入追踪平台。 +> 技术:Python 3.12+、asyncio、frozen dataclass、`inspect.signature`、pytest + FakeClock + MockTransport、临时 SQLite、真实 PG 沙箱、ruff、import-linter。 +> 基线 HEAD:`a81cc91`(分支 `feature/1.3.5-call-observability`,工作区仅 `CLAUDE.md` 既有 markdown 格式差异与未跟踪 `.pi/`,两者本计划一律不动、不暂存)。 + +本计划不涉及 `reference/` 参考实现迁移(治理主循环、Lua 限流、熔断状态机、退避公式一字不改),**保真校验不适用**。不得变更:重试预算与退避、429 免预算与 stall 算法、取消结算、限流/熔断语义、推理能力表与档位解析、缓存 key 公式、`#22`/`#24` 的调度机制。 + +## 1. 边界、授权与执行纪律 + +| 项目 | 固定边界 | +| --- | --- | +| 唯一 writer | 一工作区一 writer;父会话负责前台委派与审核派发。用户已授权本轮无需逐步请示,1.3.X 合并/发布授权沿用;跨到 1.4、新公共决策或验证豁免须停下确认 | +| 公共承诺 | 只做设计已批准的四项公共面变更:`CallStats` 导出、四响应新增字段、`TelemetryRecorder` 扩 10 参、`terminal_failure` 行;**不新增其它公共 API、不改既有字段名与位置** | +| 依赖铁律 | `ports.py`/`types.py`/`errors.py` 保持最内层,`types.py` 不 import 任何实现层;`middleware/` 只依赖端口与内核;`transports/`、`telemetry/`、`backends/` 互不依赖(import-linter 执法) | +| 取消 | `CancelledError` 永不吞没;终态写**取消优先、不 shield、不开后台任务**;快照冻结是同步动作,不 await | +| 降级方向 | recorder 写失败仍只 warning;限流/熔断后端仍 fail-closed;装配期签名不符 → **当场报错**(不是 warning) | +| 证据与秘密 | 不打印 `.env`、token、Authorization;不提交 `.pi/`、`tests/outputs/`、reference;命令输出只记安全路径、状态与退出码 | +| 证据复用 | 已发布 1.3.4 的有效证据(能力矩阵、live 轮次、发布外部验证)**直接引用不重建**;本版**不补跑未变更的模型能力矩阵**,无需新增付费调用 | + +Skill 纪律:T0 已执行 `writing-plans`;T1–T3 行为变更执行 `test-driven-development`(先失败后通过的证据须落在本会话工具输出里);T1 触及运行时数据落库,执行 `structured-logging`;每次提交执行 `commit`(英文祈使标题、无 AI 签名、显式路径暂存);T4 前执行 `requesting-code-review` 与 `verification-before-completion`,收到意见执行 `receiving-code-review`;异常先 `systematic-debugging` 定根因。 + +## 2. 文件职责与不变接缝 + +| 动作 | 精确路径 | 职责 | +| --- | --- | --- | +| 修改 | `src/polygateway/types.py` | 新增 `CallStats`、私有 `_CallContext`、`CallOperation`/`EventKind` 词表;`ChatRequest` 追加内部上下文字段;四种响应追加 `call_stats` | +| 修改 | `src/polygateway/__init__.py` | 导出 `CallStats`(`__all__` 保序插入);版本号在 T4 统一 | +| 修改 | `src/polygateway/ports.py` | `TelemetryRecorder.record_llm_call` 追加 10 个 keyword-only 无默认值参数(唯一签名事实源) | +| 修改 | `src/polygateway/telemetry/schema.py` | `SQLITE_DDL` / `PG_DDL` / `SQLITE_BACKFILL` / `_PG_BACKFILL_DECLS` / `COLUMNS` **五处**同序同增 10 列,追加在物理列末尾(`PG_BACKFILL`、`_ALTER_BY_BACKEND`、`telemetry_schema_sql` 自动派生,不手改) | +| 修改 | `src/polygateway/middleware/telemetry.py` | Emitter 构造期注入 `scope` + 装配闸;三个 emit 入口收 `operation` 与领域异常对象;诊断字段单一提取 helper;`event_kind`;终态出口 helper;`TelemetryMW` 只保留 cache_hit | +| 修改 | `src/polygateway/middleware/structured.py` | 有界错误说明常量与拼装函数改为库内可复用(重问反馈与遥测说明同一口径,数值只一份) | +| 修改 | `src/polygateway/middleware/retry.py` | transport 调用前登记一次尝试;`_emit` 传异常对象与 `operation="chat"` | +| 修改 | `src/polygateway/middleware/cache.py` | `_serialize` 剔除 `call_stats`;`_rehydrate` 显式覆盖为 `None`(防历史 dict 冒充) | +| 修改 | `src/polygateway/client.py` | 自存 `now`/`emitter`;`chat` 创建上下文、附加统计、领域失败与取消经单一终态出口 | +| 修改 | `src/polygateway/embedding.py` | `embed` 拥有上下文(空输入 0 尝试 0 遥测行);逐批传递;终态请求摘要按 §3.5 | +| 修改 | `src/polygateway/ocr.py` | 上下文在 `image` 校验通过后创建;两公开方法各自给 `operation`;错误文本保留类名前缀(经出口的显式策略参数) | +| 修改 | `src/polygateway/transports/openai_compat.py` | `_status_to_error` 增 keyword-only `operation`;三个调用点显式给值:`:512`(`embed` 非 200)传 `"embedding"`(修正历史误标)、`:527`(流式 chat)与 `:625`(非流式 chat)传 `"chat"`;`ocr_text`/`parse`/`download_result` 词表不动。**归属以本表为准**:设计 §5 把 `:527` 也叙述成 `embed()` 的分支,实测 `:527` 属 `_complete_stream`(流式 chat)、`embed` 只有 `:512` 一处;勿按设计行号把两条 chat 失败回退误标成 embedding | +| 修改 | `tests/unit/test_types.py`、`tests/unit/test_ports.py` | `CallStats`/上下文纯行为、端口签名实测(含无默认值与 keyword-only) | +| 修改 | `tests/unit/test_telemetry.py` | 列序/列数、三类行语义、诊断保真、装配闸、SQLite 新旧表、`_record_minimal` 字段字典 | +| 修改 | `tests/unit/test_retry.py`、`test_client.py`、`test_cache.py`、`test_structured.py` | 计数/计时/缓存命中/结构化耗尽终态与历史 dict 防护 | +| 修改 | `tests/unit/test_embedding.py`、`test_ocr_client.py`、`test_openai_compat.py`、`test_monkey_ocr.py` | 多批、空输入、双 HTTP 一次尝试、`exc.operation` 修正、200 失败行 | +| 修改 | `tests/unit/test_pricing.py`、`tests/unit/test_usage_source_domain.py` | **仅 Emitter 调用点机械迁移**(`TelemetryEmitter(` 补 `scope=`、`emit_*` 补 `operation=`、`emit_terminal_failure` 去 `latency_ms` 改传 `stats`);两文件的 `_MemoryRecorder` 是 `**fields` 形态,recorder 签名不动 | +| 修改 | `tests/integration/test_postgres_telemetry.py` | PG auto 追加 / manual 缺列裁剪 / 旧行 NULL / 新旧进程混写(真实 `pg_sandbox`) | +| 修改 | `README.md`、`CHANGELOG.md`、`.env.example`、`research-wiki/ARCHITECTURE.md`、`research-wiki/schemas/llm-calls.md`、`research-wiki/metrics/call-telemetry-coverage.md` | 字段数实测改写、cap 覆盖面澄清、SQL 迁移五项、终态行语义 | +| 新建(验收时) | `research-wiki/findings/2026-09-09-135-call-observability-validation.md` | 红绿、变异、失败与豁免索引,≤300 行;原始输出留 `tests/outputs/135/` | + +**不改**:`errors.py`(异常类与分类逐字不动,不加可变字段)、`middleware/admission.py`、`middleware/ratelimit.py`、`middleware/breaker.py`、`sources.py`、`thinking.py`、`providers.py`、`telemetry/sqlite.py`、`telemetry/postgres.py`(两个 recorder 靠 `**fields` + `schema.COLUMNS` 自动吃到新列,逻辑零改动)、`transports/monkey_ocr.py`(双 HTTP 仍在同一 transport 调用内,不拆)。若实现时发现必须突破本清单,先说明设计依据与最小原因交父会话核定,不顺手改。 + +## 3. 跨任务接口(可执行定义,禁止占位) + +### 3.1 统计内核(`types.py`) + +`types.py` 需新增 `import uuid`、`from collections.abc import Callable` 与 `from typing import Literal`(`Mapping`/`Any` 已在;全文现无 `Literal`)。 + +**运行时求值约束(`types.py` 无 `from __future__ import annotations`,实测 :1-17)**:`Callable` 与 `Literal` 必须是**运行时 import**,不得放进 `TYPE_CHECKING`(`_CallContext.__init__` 的函数注解与 `CallOperation` 别名都在 def/赋值时求值);且 `_CallContext` 必须**定义在 `ChatRequest`(:339)之前**,否则 `call_context: _CallContext | None` 的类注解在类创建时即 `NameError`。 + +上下文是**每调用一个实例**的单任务对象:chat 重试、结构化重问、embedding 分批都在同一任务内串行推进,故计数无需锁;严禁提升为 client 实例属性。 + +```python +CallOperation = Literal["chat", "embed", "recognize_text", "parse_layout"] +CALL_OPERATIONS: tuple[CallOperation, ...] = ("chat", "embed", "recognize_text", "parse_layout") +EventKind = Literal["attempt", "cache_hit", "terminal_failure"] +EVENT_KINDS: tuple[EventKind, ...] = ("attempt", "cache_hit", "terminal_failure") + + +@dataclass(frozen=True) +class CallStats: + """一次公开调用的统计快照;第三方合成响应的 `None` 表示未知,不得伪造 0。""" + + logical_call_id: str + attempts: int + total_latency_ms: int + + +class _CallContext: + """私有可变逻辑调用上下文:只持计数、单调时钟与终态去重位,不做 I/O。""" + + __slots__ = ("logical_call_id", "_now", "_started", "_attempts", "_terminal_claimed") + + def __init__(self, *, now: Callable[[], float]) -> None: ... + + def register_attempt(self) -> None: + """transport 调用前登记一次尝试(含免预算 429 与端口本地拒绝)。""" + + def snapshot(self) -> CallStats: + """同步冻结当前快照;绝不 await,可多次调用。""" + + def claim_terminal(self) -> bool: + """首次 True、其后 False:保证每逻辑调用至多写一条终态行。""" +``` + +`ChatRequest` 追加(末尾,`dataclasses.replace` 保留同一引用;`StructuredMW` 的 replace 已核对不重建它): + +```python + call_context: _CallContext | None = field(default=None, compare=False, repr=False) + """库内部逻辑调用上下文;`None` = 库内现场构造的请求,遥测 logical_call_id 落 NULL。""" +``` + +`LLMResponse`、`EmbeddingResponse`、`OcrTextResult`、`OcrLayoutResult` 各追加**末尾**字段: + +```python + call_stats: CallStats | None = None +``` + +### 3.2 遥测出口(`middleware/telemetry.py`) + +Emitter 是全库唯一 `record_llm_call` 调用点,本版新增的一切诊断/归因取值也只在这里定型。 + +```python +class TelemetryEmitter: + def __init__( + self, + recorder: TelemetryRecorder, + *, + scope: str, + pricing: PricingTable | None = None, + text_cap: int | None, + ) -> None: + """`scope` 构造期注入(三个 client 各一行),使三类行都带池名;不拿 source_name 顶替。 + + 同处执行装配闸 `_assert_recorder_shape(recorder)`。 + """ + + async def emit_attempt( + self, + *, + request: ChatRequest, + source: SourceConfig, + call_id: str, + latency_ms: int, + response: LLMResponse | None, + error: PolyGatewayError | str | None, + reasoning_applies: bool, + operation: CallOperation, + class_prefixed_error: bool = False, + ) -> None: + """`event_kind='attempt'`;失败异常对象在此提取 status/cause/body 四列。""" + + async def emit_cache_hit( + self, *, request: ChatRequest, response: LLMResponse, operation: CallOperation + ) -> None: + """`event_kind='cache_hit'`;attempts/total_latency_ms 列恒 NULL。""" + + async def emit_terminal_failure( + self, + *, + request: ChatRequest, + call_id: str, + error: PolyGatewayError | str, + operation: CallOperation, + stats: CallStats, + class_prefixed_error: bool = False, + ) -> None: + """`event_kind='terminal_failure'`;`latency_ms` 与 `total_latency_ms` 同取 `stats`。""" +``` + +`emit_terminal_failure` **不再收 `latency_ms`**(同一冻结快照供两列,避免双时钟微差)。诊断提取收敛为一个纯 helper,只读领域异常的既有属性,**不遍历任意对象、不猜正文**: + +```python +@dataclass(frozen=True) +class _ErrorFields: + """一行遥测的错误列;未知一律 None。""" + + error: str | None + error_type: str | None + cause_type: str | None + http_status_code: int | None + error_body: str | None + + +def _error_fields( + error: PolyGatewayError | str | None, + *, + event_kind: EventKind, + class_prefixed: bool, +) -> _ErrorFields: + """三种入参形态的唯一定型点(设计 §5/§6)。""" +``` + +判定规则(测试逐条钉死): + +| 入参 | error | error_type | cause_type / http_status_code / error_body | +| --- | --- | --- | --- | +| `None` | None | None | 全 None | +| `str`(如既有 `"cancelled"`) | 原样 | None | 全 None(**不解析字符串猜诊断**) | +| 领域异常,`event_kind='attempt'` | 见下 | `type(exc).__name__` | `type(exc.__cause__).__name__ or None` / `exc.status_code` / `summarize_body` 已产出的 `exc.body_text or None` | +| 领域异常,`event_kind='terminal_failure'` | 见下 | `type(exc).__name__` | **全 None**(C1:不搬最后一次 attempt 的状态与正文冒充整池归因) | + +error 文本:`str(exc)` 为空退回 `type(exc).__name__`;`class_prefixed=True` 时前置 `"{类名}: "`(OCR 既有 metric 归组口径);取消路径一律传字符串(attempt 行沿用既有 `"cancelled"`,终态行用同一明确取消文案),故 `error_type` 与其余三列均 NULL;`ResultInvalidError` 且为终态时并入有界结构化说明: + +```python +def _structured_detail(exc: ResultInvalidError) -> str: + """结构化耗尽的有界说明:`repair=` 至多 200 字符 + `validation=` 至多 3 条×200 字符; + **不含 raw_text**(模型正文预算已由 attempt 行的 response 列承担),两段以 ` | ` 拼接。""" +``` + +复用 `structured.py` 的既有规则所有者,不复制数值:把 `_MAX_FEEDBACK_ERRORS`/`_MAX_ERROR_CHARS` 改名为 `MAX_FEEDBACK_ERRORS`/`MAX_ERROR_CHARS`,`_format_errors` 改名 `format_bounded_errors(errors: Sequence[str]) -> str`(行为逐字不变),`StructuredMW` 与 `_structured_detail` 两个消费者共同引用。 + +装配闸(设计 §7,C3): + +```python +def _assert_recorder_shape(recorder: TelemetryRecorder) -> None: + """装配期一次 `signature.bind` 形状校验:不执行写入,只证明该形状能被接受。 + + 参数名从 `TelemetryRecorder.record_llm_call` 的协议签名派生(不手抄第四份清单), + 绑定用哨兵 `None`,不读真实请求数据。`**kwargs`(VAR_KEYWORD)自动通过; + `TypeError` → `ValueError` 装配期报错;不可 inspect(C 实现等)同样按配置错误报错。 + """ +``` + +终态唯一出口(公开边界 helper,三个 client 共用): + +```python +async def emit_terminal_once( + emitter: TelemetryEmitter | None, + *, + request: ChatRequest, + context: _CallContext, + error: PolyGatewayError | str, + operation: CallOperation, + class_prefixed_error: bool = False, +) -> None: + """去重(claim_terminal)+ 同步冻结快照 + best effort 写入。 + + `emitter is None` 或已写过 → 直接返回;写入侧异常按既有降级只落 warning; + **`CancelledError` 原样传播**(取消优先,不 shield、不开后台任务)。 + """ +``` + +`TelemetryMW` 相应收缩:删除其 `except (GatewayUnavailableError, GovernanceBackendError, SourceNotConfiguredError)` 与 `except CancelledError` 两个终态分支(改由 `GatewayClient.chat` 的边界统一写,避免两处同时写),保留 `cache_hit` 记录并传 `operation="chat"`。 + +### 3.3 `_record` 的 10 个新列取值(唯一定型点) + +| 列 | attempt | cache_hit | terminal_failure | +| --- | --- | --- | --- | +| `scope` | 构造期注入值 | 同 | 同 | +| `operation` | 调用点给定的四值之一,**绝不读 `exc.operation`** | 同 | 同 | +| `logical_call_id` | `request.call_context.logical_call_id`,上下文缺席 → NULL(不造 ID) | 同 | 同 | +| `event_kind` | `'attempt'` | `'cache_hit'` | `'terminal_failure'` | +| `http_status_code` / `cause_type` / `error_body` | 见 §3.2 表(成功行全 NULL,成功不统一填 200) | 全 NULL | 全 NULL | +| `error_type` | 该行自身错误类名 | NULL | 自身类名(`AllSourcesExhausted` / `CircuitOpenError` / `ResultInvalidError` / …) | +| `attempts` / `total_latency_ms` | NULL | NULL | `stats` 两字段 | + +其余 26 列口径逐字不变;终态行仍 `cost=None`、`usage_source='unavailable'`、token 0,不复制 attempt 的用量与费用。 + +### 3.4 schema 与端口(`telemetry/schema.py`、`ports.py`) + +10 列按**同一顺序**追加进五处(`SQLITE_DDL`、`PG_DDL`、`SQLITE_BACKFILL`、`_PG_BACKFILL_DECLS`、`COLUMNS`),全部可空、无默认值,物理位置在现有末列 `reasoning_effort` 之后: + +| 顺序 | 列名 | SQLite | Postgres | +| --- | --- | --- | --- | +| 1 | `scope` | TEXT | TEXT | +| 2 | `operation` | TEXT | TEXT | +| 3 | `logical_call_id` | TEXT | TEXT | +| 4 | `event_kind` | TEXT | TEXT | +| 5 | `http_status_code` | INTEGER | INTEGER | +| 6 | `error_type` | TEXT | TEXT | +| 7 | `cause_type` | TEXT | TEXT | +| 8 | `error_body` | TEXT | TEXT | +| 9 | `attempts` | INTEGER | INTEGER | +| 10 | `total_latency_ms` | INTEGER | INTEGER | + +`COLUMNS` 由 26 → 36,物理列(含 `created_at`)27 → 37。**不建索引、不改旧列、不回填旧行、不 ALTER 默认生产 PG**(manual 档仍裁剪写入)。`ports.py` 按同序追加 10 个 keyword-only 无默认值参数,docstring 更新为"36 字段冻结",并说明 `error_body` 沿用 `summarize_body` 上限、不进 `PGW_TELEMETRY_TEXT_CAP` 覆盖面。 + +### 3.5 三条链路的上下文所有权 + +| 链路 | 创建点 | 传递 | 附加统计 | 终态 | +| --- | --- | --- | --- | --- | +| chat | `GatewayClient.chat`:overlay/维度/档位三项校验**之后**(校验异常保持原行为,在统计边界外) | 放进 `ChatRequest.call_context`,洋葱各层经 `replace` 保留同一引用;`RetryMW._attempt` 在 `transport.complete` 前 `register_attempt()`(**上下文为 `None` 时跳过**——库内现场构造的 `ChatRequest` 不得因此报错) | 返回前 `dataclasses.replace(response, call_stats=ctx.snapshot())`(含缓存命中路径与内联遥测耗时) | `except PolyGatewayError` / `except CancelledError` 各调 `emit_terminal_once`,随后原样 `raise`;非领域异常不捕、无终态 | +| embed | `EmbeddingClient.embed`:`texts` 类型校验与**调用方维度校验**(`validate_caller_dimensions`)之后;`expected_dim` 校验在 `_attempt` 内,本就在统计边界内 | 显式参数传到 `_embed_batch` → `_attempt`;`_emit` 现场构造的 `ChatRequest` 带 `call_context=ctx`;`transport.embed` 前登记 | 合并结果 `replace(..., call_stats=ctx.snapshot())`;**空输入返回真实 `attempts=0` 统计且不写任何遥测行** | 同上,`operation="embed"`;终态 messages = `` 占位 + 第一批(至多 `batch_size` 条、每条 200 字符,与逐批行同款构造) | +| OCR | `OcrClient._call`:`image` 的 `bytes`/非空校验**通过之后**(M1 例外) | `_call` 拥有上下文并返回 `tuple[_AttemptOutcome, CallStats]`,两个公开方法据此附加统计;`transport` 调用前登记(layout 的 POST+ZIP GET 计 **1** 次) | `OcrTextResult` / `OcrLayoutResult` 的 `call_stats` | `_call` 内 `except PolyGatewayError` / `except CancelledError` 调 `emit_terminal_once`,`operation` 由公开方法传入(`recognize_text` / `parse_layout`),`class_prefixed_error=True`;`no_sources` 的 `raise` 在循环之前,`try` 必须**包住该 raise**,否则无源终态行写不出 | + +`GatewayClient` 需自存 `self._now` 与 `self._emitter`(现未保存);统计生效与否**不由 telemetry 是否启用决定**(`emitter is None` 时统计照常,仅无行可写)。 + +## 4. 任务与提交点 + +### T0:设计批准状态、计划与基线(本任务) + +- [x] 设计文档状态行改为"人类已批准(2026-09-09)",新建本计划(≤600 行)。 +- [x] 记录基线:`git status --short --branch`、`git log --oneline -3`、HEAD `a81cc91`;确认源码零差异,`CLAUDE.md` 既有 markdown 差异与 `.pi/` 一律不动、不暂存。 +- [ ] 基线验证(由执行者在 T1 之前跑一次,作为"先失败"的对照底):`make check`;`conda run -n PolyGateway pytest tests/unit -q` → 预期全绿。**基线计数以本会话实跑输出为准**(近期会话记录为 1385 项量级),不拿计划里的数字当期待值;实际失败照录,不改期待绕过。 +- [ ] 计划自审 + 独立模型审查(替代 Codex,用户已同意);意见逐条核验后就地修订。 +- [ ] wiki 登记:`.claude/tools/research_wiki.py add_entity/add_edge/rebuild_index`(design ↔ plan `implements`),登记前先确认工具不覆盖同路径已有文档。 +- [ ] 提交点:`docs: record approved call observability design and plan`(生产改动前的回滚点)。 + +### T1:统计内核与三条链路的逻辑边界(不动遥测列) + +**文件**:`types.py`、`__init__.py`、`client.py`、`embedding.py`、`ocr.py`、`middleware/retry.py`、`middleware/cache.py`;测试 `tests/unit/test_types.py`、`test_client.py`、`test_retry.py`、`test_cache.py`、`test_embedding.py`、`test_ocr_client.py`、`test_structured.py`。 + +按 §3.1 与 §3.5 实现。先写测试并确认在旧实现上红(`AttributeError: 'LLMResponse' object has no attribute 'call_stats'` 不算行为红——先落一个"同一次调用的重试次数无处可读"的行为断言,再实现)。 + +| 红绿组 | 必须证明 | +| --- | --- | +| 计数 | 一次成功=1;失败重试=实际尝试数;免预算 429 计入;多源拒绝(端口本地拒绝)计入;结构化重问计入同一上下文且不重置;embedding 三批=3;OCR layout 双 HTTP=1;缓存命中=0;空输入=0 | +| 逻辑 ID | 重试/重问/分批共享同一 `logical_call_id`;同一 client 并发两次调用互不串(`asyncio.gather` 两路各自计数与 ID);`parent_call_id` 语义不变 | +| 计时(FakeClock) | 缓存 IO、退避 sleep、准入等待、重问、内联遥测收尾全部计入 `total_latency_ms`;毫秒/秒不混用(1.5s → 1500);`emitter=None` 时统计仍正确。**替身构造要求**:假缓存后端的 `get`/`set` 与假 recorder 的 `record_llm_call` 内部**必须主动推进 FakeClock**,否则"缓存 IO/内联遥测计入总耗时"的断言会退化成恒等于 0 的空转绿 | +| 缓存不毒化 | `_serialize` 输出无 `call_stats` 键;手写含 `"call_stats": {...}` 的历史条目命中后 `response.call_stats is None`(**dict 不得冒充 `CallStats`**);缓存 key 逐字节不变(黄金值) | +| 附加点 | 失败异常上**不附任何统计字段**(`hasattr(exc, "call_stats") is False`);`CancelledError` 类型与语义不变;非领域异常原样传播 | +| 空输入 | `embed([])` 返回 `attempts=0`、真实 `logical_call_id`,且注入的内存 recorder **零行** | + +**验证**:`conda run -n PolyGateway pytest tests/unit/test_types.py tests/unit/test_client.py tests/unit/test_retry.py tests/unit/test_cache.py tests/unit/test_embedding.py tests/unit/test_ocr_client.py tests/unit/test_structured.py -q` → 目标断言先红后绿,其余保留行为绿;`make check`(含 import-linter:`types.py` 不得引入实现层 import)。 + +- [ ] 提交点:`feat: track logical call statistics across governed calls` + +### T2:遥测 10 列、诊断保真、scope/operation 与装配闸 + +**文件**:`telemetry/schema.py`、`ports.py`、`middleware/telemetry.py`、`middleware/structured.py`、`middleware/retry.py`、`transports/openai_compat.py`、三个 client 的 Emitter 构造行;测试 `tests/unit/test_telemetry.py`、`test_ports.py`、`test_openai_compat.py`、`test_monkey_ocr.py`、`test_structured.py`。 + +按 §3.2–§3.4 实现(本任务只产出 attempt / cache_hit 两类行的新列,终态行留 T3)。先写测试确认旧实现红:现状下 `error` 列是被 `str()` 压平的自由文本、无 `scope`/`operation` 列、旧签名 recorder 只落 warning。 + +| 红绿组 | 必须证明 | +| --- | --- | +| 列与 SQL | `len(COLUMNS) == 36`、物理列 37、五处列序一致(新建库与 ALTER 追加列序相同);`insert_sql` 两端语句更新;`telemetry_schema_sql` 输出与库内 DDL 同源 | +| 端口 | `inspect.signature` 实测 10 个新参存在、keyword-only、无默认值;docstring 字段数与实测一致 | +| 诊断保真 | 中转把 529 改写成 503 → 记 503(**不猜回 529**);直接 529 → 529;空 `str()` 的 Connect/Read/Write/PoolTimeout → error 落类名、`cause_type` 为对应 httpx 类名;`error_body` 为 `summarize_body` 摘要且不等于 `raw_text`;成功行五列 NULL(不填 200) | +| operation | `embed` 非 200 → `exc.operation == "embedding"`(历史误标修正);新列 `operation` 恒为四值之一,与 `exc.operation` 无关、不随异常变化;`monkey_ocr` 的 `success != true` 带 **200** 的失败行 → `http_status_code == 200`(该列不可作失败判据) | +| scope | attempt 与 cache_hit 行都带 scope;model/provider/source 未选出时仍留原空值 | +| 字符串入参 | 取消 attempt 的 `"cancelled"` 仍原样落 `error`,其余四列 NULL(**不解析字符串**) | +| 装配闸 | 旧签名 recorder → 构造 `TelemetryEmitter`(即三个 client 装配)**抛错**而非 warning;`**kwargs` recorder 通过;不可 inspect 的对象 → 配置错误报错;参数名确由协议签名派生(改协议后闸自动跟随的断言) | +| 有界说明 | `format_bounded_errors` 重命名后 `StructuredMW` 反馈文案逐字不变(黄金串);常量只有一份定义 | + +**验证**:`conda run -n PolyGateway pytest tests/unit/test_telemetry.py tests/unit/test_ports.py tests/unit/test_openai_compat.py tests/unit/test_monkey_ocr.py tests/unit/test_structured.py -q`;`make check`。 + +- [ ] 提交点:`feat: record scope, operation and failure diagnostics per row` + +### T3:终态行、取消口径与统一出口 + +**文件**:`middleware/telemetry.py`(终态 helper 与 `TelemetryMW` 收缩)、`client.py`、`embedding.py`、`ocr.py`;测试 `tests/unit/test_telemetry.py`、`test_client.py`、`test_embedding.py`、`test_ocr_client.py`、`test_structured.py`、`test_retry.py`。 + +按 §3.2 的 `emit_terminal_once` 与 §3.5 的三条链路实现。先写测试确认旧实现红:结构化耗尽当前**没有任何失败行**;embedding/OCR 的无源、准入拒绝、重试耗尽与尝试外取消同样无终态行。 + +| 红绿组 | 必须证明 | +| --- | --- | +| 补漏 | chat 结构化耗尽、embedding/OCR 的 `no_sources`、准入拒绝、`retry_exhausted`、尝试外取消各恰有 **1** 条 `terminal_failure` | +| 不变量 I3 | 领域失败每调用至多 1 条(recorder 写失败仅 warning,SQL 可见 ≤1);重复调用出口不产生第二条(`claim_terminal`);直接 `RequestRejectedError` / `ResultInvalidError` 现在**既有 attempt 错误行也有终态行**(400 密集负载错误行翻倍是已批准的下游可见变化) | +| 非领域异常 | 编程错(如 `KeyError`)→ **0 条**终态行、原样传播、分类不被改写 | +| 取消三路 | chat / embed / OCR 同策略尽力写一条(允许 0);**终态写的 await 上被取消 → `CancelledError` 传播**(不 shield、无后台任务);permit 与探针释放行为不变(`inflight == 0`) | +| 归因 SQL | 设计 §5 那条 `WHERE logical_call_id = :lcid` 查询同时给出终态 reason 文案与逐源状态码/正文;终态行 `http_status_code`/`cause_type`/`error_body` 三列 NULL;结构化耗尽的终态 `error` 含有界 validation/repair 说明且**不含 `raw_text`** | +| 行语义 | 三类行均带 scope;`event_kind` 可区分;`SUM(cost)` 不因终态行变化(终态 `cost IS NULL`、usage `unavailable`);`AVG(latency_ms)` 按 `event_kind` 分组的断言;终态 `latency_ms == total_latency_ms`(同一快照) | +| 快照时机 | 终态快照不含自身写入耗时(FakeClock:写入内推进时钟,列值不变);成功响应快照含返回前已完成的内联遥测耗时 | +| 摘要 | embedding 终态 messages 为占位 + 第一批截断文本,不含全量原输入;OCR 终态沿用 ``,图像 bytes 不入库;OCR 终态 error 保留类名前缀 | + +**验证**:`conda run -n PolyGateway pytest tests/unit -q`(全量单测,含上述文件);`make check`。 + +- [ ] 提交点:`feat: emit one terminal failure row per logical call` + +### T4:旧测试机械核对、存储兼容、变异证据、文档与独立验证 + +**文件**:下述机械核对清单 + `tests/integration/test_postgres_telemetry.py` + 文档清单 + 验收 finding。 + +**一次机械核对(禁止分散反复修)**:本版真正的破坏面**不只是 recorder 签名,还有 Emitter 侧**(`TelemetryEmitter.__init__` 新增必填 `scope`、三个 `emit_*` 新增必填 `operation`、`emit_terminal_failure` 去 `latency_ms` 改收 `stats`),这些关键词不含 `record_llm_call`/`COLUMNS`。故全量清单用: + +```bash +grep -rn "record_llm_call\|COLUMNS\|_EXPECTED_COLUMNS\|TelemetryEmitter(\|emit_attempt(\|emit_cache_hit(\|emit_terminal_failure(" tests/ --include=*.py +``` + +一轮改完再跑,不逐个文件试错。已核实的免改项:`tests/unit/test_backpressure.py:231` 的假 emitter 是 `emit_attempt(self, *args, **kwargs)`,兼容;`tests/integration/` 无 Emitter 构造点。已知点: + +| 位置 | 动作 | +| --- | --- | +| `tests/unit/test_ports.py:99` `_DummyRecorder` | 显式签名补齐 10 参(它是"新签名可实现"的活证据,不改成 `**kwargs`) | +| `tests/unit/test_ports.py:264` `TestTelemetryRecorderSignature` | 新参进 `no default` / keyword-only 参数化 | +| `tests/unit/test_telemetry.py:40` `_EXPECTED_COLUMNS`、:220/:321/:597/:639/:647/:686-698 计数与尾部断言 | 26→36、27→37、尾部 10 列、旧表 backfill 目标列数 | +| `tests/unit/test_telemetry.py:108` `_record_minimal` | 字段字典补 10 键(默认 NULL 形态) | +| `tests/integration/test_postgres_telemetry.py:100` 字段字典 | 同上(该函数返回值被逐列断言消费,改动须与 `COLUMNS` 同序) | +| `test_client.py` / `test_embedding.py` / `test_ocr_client.py` / `test_openai_compat.py` / `test_pricing.py` / `test_cache.py` / `test_usage_source_domain.py` 的 `_MemoryRecorder` | 均为 `**fields` 形态,**recorder 签名无需改**;只需核对断言里的字段计数与新列期望 | +| `tests/unit/test_pricing.py`、`tests/unit/test_usage_source_domain.py` 的 **Emitter 调用点** | **须改**:`TelemetryEmitter(` 补 `scope=`、`emit_attempt(`/`emit_cache_hit(` 补 `operation=`;`test_usage_source_domain.py:297` 的 `emit_terminal_failure(...)` 同时缺 `scope`/`operation`/`stats` 且多传 `latency_ms`,不改必 `TypeError`(recorder 形态兼容 ≠ emitter 调用点兼容) | + +**存储兼容**(真实 PG,复用 `tests/integration/conftest.py` 的 `pg_sandbox` / `pg_catalog_probe`,不新建沙箱设施;**不引用 `assert_no_leftovers`**——它是 `tests/integration/test_pg_sandbox.py:19` 的模块级 fixture,对 `test_postgres_telemetry.py` 不可见,上提它要改 §2 清单外的 `conftest.py`。本版不新增沙箱资源创建路径,残留风险与 1.3.4 逐字相同,由该文件既有用例覆盖,属可复用的既有证据): + +- SQLite:新建库 37 列;旧表(1.3.4 形态 27 列)auto 档补齐 10 列且列序与新建库一致;manual 档不发 DDL、按现有列裁剪写入并发一条点名缺列的 warning;旧行新列为 NULL。 +- PG(验收取以下**四项**):auto 追加 10 列;**manual 缺列裁剪**写入成功且不抛(`_trim_columns` 路径);旧行新列为 NULL;新旧进程混写同一表(旧列集写入 + 新列集写入并存)。 + +**变异证据**(仓库外临时副本 + `PYTHONPATH=<副本>/src`,先确认 `polygateway.__file__` 指向副本;逐个变异 → 跑指定节点记 exit 1 与被杀断言 → 恢复校验散列 → exit 0;**绝不在主工作区改生产代码凑红**): + +| 变异 | 必须被杀死的断言 | +| --- | --- | +| `register_attempt()` 移到 transport 调用之后的 `except` 分支外/内错位 | 失败重试与 429 计数断言 | +| `ChatRequest` 上下文字段改为 `replace` 时新建实例(模拟上下文复制) | 重问/分批共享同一 `logical_call_id` 的断言 | +| `_rehydrate` 去掉 `call_stats=None` 覆盖 | 历史 dict 冒充 `CallStats` 的断言 | +| Emitter 入口提前 `str(exc)` 压平 | `error_type`/`http_status_code`/`error_body` 保真断言 | +| 终态行复制最后一次 attempt 的 token/cost | 费用聚合与终态 `cost IS NULL` 断言 | +| 去掉 `claim_terminal` 去重 | 每失败调用至多一条终态的断言 | +| 装配闸改为捕获 `TypeError` 后 warning | 旧签名 recorder 装配期报错的断言 | + +**文档同步(发布前必须同批;wiki 站点自 2026-08-02 下线,按 docs-convention §2 的下线期条款,承接方为 README / CHANGELOG / .env.example / ARCHITECTURE 四处)**: + +| 位置 | 改什么 | +| --- | --- | +| `README.md:23` | "必录 26 字段" → 实测值(`len(inspect.signature(TelemetryRecorder.record_llm_call).parameters) - 1`,预期 36),不凭记忆 | +| `README.md:26/399/455`、`ARCHITECTURE.md:592`、`.env.example:109` | `PGW_TELEMETRY_TEXT_CAP` 覆盖面仍是四处;明确 `error_body` 沿用 `summarize_body` 上限、`error` 的新增结构化说明另有独立限长,**二者都不在 cap 覆盖内** | +| `README.md` 能力表 + 新增小节 | 四响应的 `call_stats` 读法;设计 §8 的 SQL 迁移**五项**(失败行数改判据、`error IS NOT NULL` 不再是判据、`AVG(latency_ms)` 须按 `event_kind` 分组、费用口径不变、失败行可能带 200) | +| `ARCHITECTURE.md:565` 必录字段清单 + §7.8 补列一节 | 追加 10 列语义、`event_kind` 三态、终态行不变量 I3/I4、`operation` 与 `exc.operation` 是两个语义 | +| `CHANGELOG.md` 未发布段 | 公共面四项 + 下游动作清单(点名"计失败调用改 `WHERE event_kind = 'terminal_failure'`"与自定义 recorder 的装配期报错) | +| `research-wiki/schemas/llm-calls.md`、`metrics/call-telemetry-coverage.md` | 复用既有实体登记新列与三类行口径;真实覆盖基线待首次运行填,不写伪百分比 | + +**验收命令与证据**: + +| 检查 | 命令 / 要求 | +| --- | --- | +| 静态 | `make check`;`git diff --check`;`conda run -n PolyGateway python -m compileall -q src/polygateway` | +| 日常全量 | `make test`(真实退出码、coverage ≥80%,连接依赖 skip 单列) | +| 集成 | `conda run -n PolyGateway pytest tests/integration/test_postgres_telemetry.py tests/integration/test_governance_stack.py -q` | +| 契约 | `conda run -n PolyGateway pytest tests/contracts -q`(限流契约随实现一起交付的既有套件) | +| 变异 | 上表七项逐条 exit 1 → 还原 exit 0,日志留 `tests/outputs/135/` | +| 独立验证 | 由**父会话前台派全新上下文 verifier**:只给批准设计、本计划、分支 diff 与命令,不给实现自评;至少覆盖"正确性/回归"与"下游可见变化/范围"两角度,Critical/Important 清零 | +| slow | `pytest -m slow` 属发布清单第 4 步(CLAUDE §4.4.1),在合并 main 之后统一跑;**本版不新增 live 轮次、不补跑未变更的模型能力矩阵**,1.3.4 已有有效证据直接引用 | + +长跑用 tmux,`PYTHONUNBUFFERED=1`,输出重定向到 `tests/outputs/135/`,命令后**不得接管道**(退出码失真),完成判定用 `wait`/PID 轮询,不用会自匹配的 `pgrep -f`。 + +- [ ] 提交点:`docs: document logical call telemetry and migration impact`(机械核对与存储兼容若需单独回滚点,可先提 `test: align telemetry fixtures with the 36-field recorder`,仍在 5 个提交内) + +## 5. 测试矩阵 → 任务映射(设计 §9 逐族落位) + +| 设计测试族 | 任务 | +| --- | --- | +| logical 计数(含空输入 0 尝试 0 行) | T1 | +| 计时(缓存/退避/准入/重问/收尾、关 recorder) | T1(快照时机的终态部分在 T3) | +| 失败与取消(三路同策略、终态 await 取消传播、非领域异常 0 行) | T3 | +| 保真诊断(529/503、空超时文案、`exc.operation`、新列四值) | T2 | +| 归因 SQL(终态 reason + 逐源现场、结构化说明无 `raw_text`) | T3 | +| 行语义(三类行 scope、至多一条终态、费用不重复、`AVG` 分组) | T2(前两项)+ T3 | +| 装配闸(旧签名报错、`**kwargs` 通过、不可 inspect) | T2 | +| 存储兼容(SQLite 新旧表、PG manual/auto、旧行 NULL、混写) | T4 | +| 变异(计数位置、上下文复制、缓存回放、提前压平、终态双计费用) | T4 | + +**共享测试设施一律复用,不新建**:`tests/contracts/conftest.py:26` 与 `tests/unit/test_ocr_client.py:344` 的 `FakeClock`(注入 `now`/`sleep`,确定性计时);各测试文件既有的 `_MemoryRecorder`(`**fields`)与脚本化 MockTransport;`tmp_path` + 真实 `SQLiteRecorder(auto_migrate=...)`;`tests/integration/conftest.py` 的 PG 沙箱三件套。真实 LLM 输出如产生,按 CLAUDE §4.6 落 `tests/outputs//`。 + +## 6. 阻塞矩阵与发布交接 + +| 情形 | 本库可完成 | 不可自行宣称/处置 | +| --- | --- | --- | +| 无 PG 可用 | 单测与 SQLite 全部覆盖;PG 用例 skip 并单列 | 不得把 skip 记作通过;发布前须补跑或取人类具名豁免 | +| 下游自定义 recorder 未知 | 装配闸 + 迁移文档 + `**kwargs` 兼容路径 | 不能声称"后端完全不受影响":其 schema、INSERT 字段与契约测试仍须同步 | +| 设计外漏洞 | 独立记录实际文件与反例,交父会话核定 | 不顺手实施 #20 之外的 issue、不新增表/端口/deadline/hedging | +| 发布 | 门全绿后按 CLAUDE §4.4.1 逐步执行(文档先行 → CHANGELOG 定版 → 双处版本号 → 合并 main → `make lint`/`make test`/`pytest -m slow` → tag → 构建 → 上传 → 下载验证 → Release + 挂仓库 + 页面核对) | 本计划**不复制**该清单,也不预先勾选任何发布步骤;只 bump 版本号不叫发布 | + +## 7. 自审 + +- 设计每节可指到任务:§3→T1、§4→T1、§5→T2/T3、§6→T3、§7→T2、§8→T4、§9→§5 映射表、§10 六项批准项全部落在 T1–T3 的公共面改动内。 +- 无占位符与待定项:跨任务消费的类型(`CallStats`、`_CallContext`、`_ErrorFields`)、四个 emit 入口、两个 helper(`_assert_recorder_shape`、`emit_terminal_once`)、10 列取值表、10 列 DDL 类型均已写出可执行定义;被引用的 `PolyGatewayError`、`ResultInvalidError`、`summarize_body`、`format_bounded_errors`、`FakeClock`、`pg_sandbox` 全部指向既有实现或本计划已定义项。 +- 一致性核对:`emit_terminal_failure` 去掉 `latency_ms` 后,唯一调用者是 `emit_terminal_once`;`TelemetryMW` 的终态分支删除后 chat 终态只剩客户端边界一处;`operation` 只由调用点给定,链路上无任何位置读 `exc.operation`。 +- 未采纳项(异常上挂可变统计、终态搬运最后一次 attempt 的状态与正文、成功侧汇总行)理由在设计正文,本计划不复活。 +- 本计划编写过程**未运行 pytest、未做变异、未调用任何模型、未提交**;T0 基线验证与其后各任务的红绿证据由执行者在自己的会话内出示。 From 87c261bf7353cd9438edff54d2861b1b55a43254 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 10:04:39 -0400 Subject: [PATCH 05/10] feat: track logical call statistics across governed calls --- src/polygateway/__init__.py | 2 + src/polygateway/client.py | 13 ++- src/polygateway/embedding.py | 17 +++- src/polygateway/middleware/cache.py | 7 ++ src/polygateway/middleware/retry.py | 5 + src/polygateway/ocr.py | 28 +++++- src/polygateway/types.py | 109 ++++++++++++++++++++- tests/unit/test_cache.py | 72 ++++++++++++++ tests/unit/test_client.py | 137 +++++++++++++++++++++++++++ tests/unit/test_embedding.py | 40 ++++++++ tests/unit/test_ocr_client.py | 31 ++++++ tests/unit/test_retry.py | 58 ++++++++++++ tests/unit/test_types.py | 141 ++++++++++++++++++++++++++++ 13 files changed, 650 insertions(+), 10 deletions(-) diff --git a/src/polygateway/__init__.py b/src/polygateway/__init__.py index 6254135..4043c5a 100644 --- a/src/polygateway/__init__.py +++ b/src/polygateway/__init__.py @@ -39,6 +39,7 @@ from polygateway.thinking import ( ) from polygateway.types import ( EFFORT_ORDER, + CallStats, Effort, EmbeddingResponse, LLMResponse, @@ -57,6 +58,7 @@ __all__ = [ "EFFORT_ORDER", "Effort", "AllSourcesExhausted", + "CallStats", "CircuitOpenError", "EmbeddingClient", "EmbeddingResponse", diff --git a/src/polygateway/client.py b/src/polygateway/client.py index c5f347c..9ac1a64 100644 --- a/src/polygateway/client.py +++ b/src/polygateway/client.py @@ -9,6 +9,7 @@ from __future__ import annotations import asyncio +import dataclasses import hashlib import json import random @@ -46,6 +47,7 @@ from polygateway.types import ( Effort, LLMResponse, TelemetryStatus, + _CallContext, coerce_effort, validate_caller_dimensions, validate_request_overlay, @@ -289,6 +291,10 @@ class GatewayClient: self._structured_available = structured_strategy is not None self._terminal = terminal # 内部引用: 装配自省/测试用 self._handler = compose(middlewares, terminal) + # 逻辑调用统计需要同一只注入钟(1.3.5);现之前只传给中间件未自存 + self._now = now + # 终态行由公开边界统一写出(T3),故边界也需持有 emitter + self._emitter = emitter self._transport = transport self._telemetry = telemetry self._cache = cache @@ -370,6 +376,8 @@ class GatewayClient: else coerce_effort(reasoning_effort, origin="chat(reasoning_effort=...)") ) validate_thinking_raw(sampling, effort=effort, wire=None, origin="chat overlay") + # 三项校验均已通过 → 进入统计边界(设计 §3: 输入校验异常在边界之外,保持原行为) + context = _CallContext(now=self._now) request = ChatRequest( messages=messages, session_id=session_id, @@ -383,8 +391,11 @@ class GatewayClient: reasoning_effort=effort, tenant_id=dimension_tenant_id, meta=dimensions, + call_context=context, ) - return await self._handler(request) + response = await self._handler(request) + # 快照在返回前冻结: 故它含缓存命中路径与已完成的内联遥测耗时 + return dataclasses.replace(response, call_stats=context.snapshot()) async def aclose(self) -> None: """幂等释放**自建**资源: transport、遥测、缓存、限流/熔断后端。 diff --git a/src/polygateway/embedding.py b/src/polygateway/embedding.py index ac042ca..44ac701 100644 --- a/src/polygateway/embedding.py +++ b/src/polygateway/embedding.py @@ -16,6 +16,7 @@ from __future__ import annotations import asyncio +import dataclasses import math import random import time @@ -47,6 +48,7 @@ from polygateway.types import ( EmbeddingResponse, LLMResponse, TelemetryStatus, + _CallContext, strip_unsupported_extra_body, validate_caller_dimensions, ) @@ -184,7 +186,11 @@ class EmbeddingClient: dimension_tenant_id, dimensions = validate_caller_dimensions( tenant_id, meta, origin="embed(tenant_id=..., meta=...)" ) + # 校验均已通过 → 进入统计边界(设计 §3.5: `texts` 类型与调用方维度校验之后) + context = _CallContext(now=self._now) if not texts: + # 合法零尝试: 返回真实统计(attempts=0),且**不写任何遥测行** + # ——与 cache_hit 不同,不要按"遥测必录"推断它有台账行(设计 §3 M2) return EmbeddingResponse( vectors=[], dim=0, @@ -195,6 +201,7 @@ class EmbeddingClient: latency_ms=0, call_id=str(uuid.uuid4()), source_name="", + call_stats=context.snapshot(), ) if not self._sources: raise AllSourcesExhausted(scope=self._scope, reason="no_sources", retry_after_s=0.0) @@ -207,9 +214,11 @@ class EmbeddingClient: parent_call_id, dimension_tenant_id, dimensions, + context, ) ) - return self._merge(outcomes) + # 全批共享同一上下文,故分批是实现细节而非 N 次独立逻辑调用 + return dataclasses.replace(self._merge(outcomes), call_stats=context.snapshot()) # —— 治理循环(与 RetryMW 同构;设计 §7.1 已声明的有限重复)—— @@ -220,6 +229,7 @@ class EmbeddingClient: parent_call_id: str | None, tenant_id: str | None, meta: dict[str, Any], + context: _CallContext, ) -> _BatchOutcome: fails = 0 reasons: dict[str, str] = {} @@ -232,7 +242,7 @@ class EmbeddingClient: continue async with clock.attempting(): outcome = await self._attempt( - batch, *picked, reasons, session_id, parent_call_id, tenant_id, meta + batch, *picked, reasons, session_id, parent_call_id, tenant_id, meta, context ) if isinstance(outcome, _BatchOutcome): return outcome @@ -258,10 +268,13 @@ class EmbeddingClient: parent_call_id: str | None, tenant_id: str | None, meta: dict[str, Any], + context: _CallContext, ) -> _BatchOutcome | _FailedBatch: call_id = str(uuid.uuid4()) started = self._now() actual = 0 + # 登记在 transport 调用**之前**(同 RetryMW): 失败与取消的尝试也真的发出去了 + context.register_attempt() try: result = await self._transport.embed(texts=batch, source=source, call_id=call_id) if self._expected_dim is not None and result.dim != self._expected_dim: diff --git a/src/polygateway/middleware/cache.py b/src/polygateway/middleware/cache.py index abdc2e7..af60be4 100644 --- a/src/polygateway/middleware/cache.py +++ b/src/polygateway/middleware/cache.py @@ -208,6 +208,10 @@ class CacheMW: max_inter_token_ms=None, call_id=str(uuid.uuid4()), structured_data=structured_data, + # 显式覆盖: 历史条目里的 `call_stats` 是个 dict,而 `_RESPONSE_FIELDS` + # 过滤**会放行它**——不覆盖就会有 dict 冒充 `CallStats` 漏给调用方。 + # 本次调用的真实统计由公开边界在返回前追加(设计 §3) + call_stats=None, ) return LLMResponse(**fields) except Exception as exc: @@ -230,6 +234,9 @@ class CacheMW: def _serialize(self, response: LLMResponse) -> str: data = dataclasses.asdict(response) data.pop("structured_data", None) # pydantic 实例不可 JSON 往返(设计 §2.1) + # 统计描述**本次**调用,存进去再放出来等于向下一个调用方谎称 + # 它重试了 N 次;`asdict` 会把 `CallStats` 摊成 dict,故必须显式剔除 + data.pop("call_stats", None) return json.dumps(data, ensure_ascii=False) async def _safe_get(self, key: str) -> str | None: diff --git a/src/polygateway/middleware/retry.py b/src/polygateway/middleware/retry.py index 09a0236..0a3a1ea 100644 --- a/src/polygateway/middleware/retry.py +++ b/src/polygateway/middleware/retry.py @@ -278,6 +278,11 @@ class RetryMW: call_id = str(uuid.uuid4()) started = self._now() actual = 0 + # 登记在 transport 调用**之前**(1.3.5 设计 §4): 失败与取消的尝试同样 + # "真的打出去了",挪到成功之后会让诊断最需要看见的那几次从计数里消失。 + # 上下文为 None = 库内现场构造的请求,跳过而不是报错 + if request.call_context is not None: + request.call_context.register_attempt() try: result = await self._transport.complete( messages=request.messages, diff --git a/src/polygateway/ocr.py b/src/polygateway/ocr.py index 7dc0d35..683649f 100644 --- a/src/polygateway/ocr.py +++ b/src/polygateway/ocr.py @@ -40,12 +40,14 @@ from polygateway.middleware.retry import StallClock, _failure_reason, backoff_de from polygateway.middleware.telemetry import TelemetryEmitter from polygateway.ports import OutcomeAwareSelector from polygateway.types import ( + CallStats, ChatRequest, LLMResponse, OcrLayoutResult, OcrTextResult, TelemetryStatus, Usage, + _CallContext, strip_unsupported_extra_body, validate_caller_dimensions, ) @@ -176,7 +178,7 @@ class OcrClient: dimension_tenant_id, dimensions = validate_caller_dimensions( tenant_id, meta, origin="recognize_text(tenant_id=..., meta=...)" ) - outcome = await self._call( + outcome, call_stats = await self._call( "text", image, session_id, parent_call_id, dimension_tenant_id, dimensions ) result = outcome.result @@ -187,6 +189,7 @@ class OcrClient: latency_ms=outcome.latency_ms, call_id=outcome.call_id, raw=result.raw, + call_stats=call_stats, ) async def parse_layout( @@ -206,7 +209,7 @@ class OcrClient: dimension_tenant_id, dimensions = validate_caller_dimensions( tenant_id, meta, origin="parse_layout(tenant_id=..., meta=...)" ) - outcome = await self._call( + outcome, call_stats = await self._call( "layout", image, session_id, parent_call_id, dimension_tenant_id, dimensions ) result = outcome.result @@ -218,6 +221,7 @@ class OcrClient: latency_ms=outcome.latency_ms, call_id=outcome.call_id, raw=result.raw, + call_stats=call_stats, ) async def check_health(self) -> dict[str, bool]: @@ -242,11 +246,14 @@ class OcrClient: parent_call_id: str | None, tenant_id: str | None, meta: dict[str, Any], - ) -> _AttemptOutcome: + ) -> tuple[_AttemptOutcome, CallStats]: if not isinstance(image, bytes): raise TypeError("image 必须是 bytes(路径读取/批量拼帧留业务侧,D9)") if not image: raise ValueError("image 不能为空") + # M1 例外: `image` 校验在 `_call` 内而非公开方法,故上下文在该校验 + # **通过之后**创建——这样设计 §3 的"校验在统计边界外"对 OCR 才成立 + context = _CallContext(now=self._now) if not self._sources: raise AllSourcesExhausted(scope=self._scope, reason="no_sources", retry_after_s=0.0) fails = 0 @@ -260,10 +267,18 @@ class OcrClient: continue async with clock.attempting(): outcome = await self._attempt( - kind, image, *picked, reasons, session_id, parent_call_id, tenant_id, meta + kind, + image, + *picked, + reasons, + session_id, + parent_call_id, + tenant_id, + meta, + context, ) if isinstance(outcome, _AttemptOutcome): - return outcome + return outcome, context.snapshot() fails += 1 if fails >= self._retry.max_attempts: raise AllSourcesExhausted( @@ -287,11 +302,14 @@ class OcrClient: parent_call_id: str | None, tenant_id: str | None, meta: dict[str, Any], + context: _CallContext, ) -> _AttemptOutcome | _FailedAttempt: call_id = str(uuid.uuid4()) started = self._now() # 四个 emit 分支(成功/终态拒绝/取消/可重试失败)都必须带调用方维度: # 失败行与取消行同样需要租户归属,漏掉任一分支就会写出无归属的行 + # layout 的 POST + ZIP GET 在同一次 `_invoke` 内,故这里只登记 **1** 次 + context.register_attempt() try: result = await self._invoke(kind, image, source, call_id) await self._record_quietly(self._breaker.record_success(entry)) diff --git a/src/polygateway/types.py b/src/polygateway/types.py index 2857d8e..8800036 100644 --- a/src/polygateway/types.py +++ b/src/polygateway/types.py @@ -8,11 +8,12 @@ import dataclasses import json import math import re -from collections.abc import Mapping +import uuid +from collections.abc import Callable, Mapping from dataclasses import dataclass, field from enum import StrEnum from types import MappingProxyType -from typing import Any +from typing import Any, Literal from loguru import logger @@ -274,6 +275,91 @@ class ThinkingObservation(StrEnum): UNKNOWN = "unknown" +CallOperation = Literal["chat", "embed", "recognize_text", "parse_layout"] +"""遥测 `operation` 列的值域: **公开方法**四值,由调用点给定。 + +与 `PolyGatewayError.operation`(HTTP 子操作,如 `download_result`)是**两个语义**, +不做自动转换;链路上任何位置都不得读 `exc.operation` 来填本列(设计 §5 I1/I2)。""" + +CALL_OPERATIONS: tuple[CallOperation, ...] = ("chat", "embed", "recognize_text", "parse_layout") + +EventKind = Literal["attempt", "cache_hit", "terminal_failure"] +"""一行遥测描述的事件形态;旧行 NULL,不回填。 + +终态行与 attempt 行**不是重复事实**(前者描述逻辑终态,后者描述单次尝试), +故禁止按 `error IS NOT NULL` 跨两类直接计失败调用次数(设计 §6/§8)。""" + +EVENT_KINDS: tuple[EventKind, ...] = ("attempt", "cache_hit", "terminal_failure") + + +@dataclass(frozen=True) +class CallStats: + """一次**公开调用**(而非单次尝试)的统计快照(设计 §3)。 + + 四种响应各平铺三字段会立刻漂移,故收敛成单一对象并由包根导出。 + 第三方合成响应的 `None` 表示**未知**,不得伪造 0。 + """ + + logical_call_id: str + """每次公开调用一个 UUID;重试、结构化重问、embedding 分批共享同一个。 + + 不占用既有 `parent_call_id`(后者是调用方的业务关联,语义不变)。""" + + attempts: int + """准入后实际调用 transport 端口的次数;含免预算 429 与端口本地拒绝。 + + **不是 HTTP 请求条数**: OCR layout 的 POST + ZIP GET 在同一次 transport + 调用内,计 1 次。缓存命中与空输入是合法的零尝试。""" + + total_latency_ms: int + """从输入校验通过到返回/异常传播前的单调时钟快照。 + + 含缓存 IO、退避等待、准入等待、重问、分批与内联记账。 + "总耗时减最后一次尝试耗时"**不等于**纯等待(含其他本地工作)。""" + + +class _CallContext: + """私有可变逻辑调用上下文: 只持计数、单调时钟与终态去重位,不做 I/O。 + + **每调用一个实例**的单任务对象: chat 重试、结构化重问、embedding 分批 + 都在同一任务内串行推进,故计数无需锁。**严禁提升为 client 实例属性** + ——那会让同一 client 的并发调用互相串掉计数与逻辑 ID(库铁律"纯 asyncio 中立"、 + VT `evolve_llm = llm` 教训的同一形态)。 + """ + + __slots__ = ("_attempts", "_now", "_started", "_terminal_claimed", "logical_call_id") + + def __init__(self, *, now: Callable[[], float]) -> None: + self.logical_call_id = str(uuid.uuid4()) + self._now = now + self._started = now() + self._attempts = 0 + self._terminal_claimed = False + + def register_attempt(self) -> None: + """transport 调用**前**登记一次尝试(含免预算 429 与端口本地拒绝)。 + + 登记点在调用前而非成功后: 否则失败与取消的尝试会从计数里消失, + 而那正是诊断时最需要看见的那几次。 + """ + self._attempts += 1 + + def snapshot(self) -> CallStats: + """同步冻结当前快照;**绝不 await**,可多次调用。""" + return CallStats( + logical_call_id=self.logical_call_id, + attempts=self._attempts, + total_latency_ms=int((self._now() - self._started) * 1000), + ) + + def claim_terminal(self) -> bool: + """首次 `True`、其后 `False`: 保证每逻辑调用至多写一条终态行。""" + if self._terminal_claimed: + return False + self._terminal_claimed = True + return True + + @dataclass(frozen=True) class LLMResponse: """一次治理调用的统一响应(与三项目超集兼容,ARCH §5.1)。""" @@ -336,6 +422,9 @@ class LLMResponse: `None` 不是"没推理": 库不表态时也不推定模型自己的默认档——"没看见"不许说成 "发生了"(同 `thinking_observation` 的 `UNKNOWN` 一脉)。""" + call_stats: CallStats | None = None + """本次**逻辑调用**的统计快照(1.3.5);`None` = 未知,不得读成 0。""" + @dataclass(frozen=True) class ChatRequest: @@ -382,6 +471,16 @@ class ChatRequest: 而档位要经能力表校验、要进缓存 key、要落遥测——混进直通层等于放弃这三样, 正是 issue #20 里下游手写 `extra_body` 绕过全部治理的那条路。""" + # —— 库内部逻辑调用上下文(1.3.5;追加在末尾,不扰动既有字段的位置构造)—— + call_context: _CallContext | None = field(default=None, compare=False, repr=False) + """库内部逻辑调用上下文;`None` = 库内现场构造的请求,遥测 `logical_call_id` 落 NULL。 + + `compare=False, repr=False` 不是洁癖: 进 `compare` 会让两个内容相同的请求因 + "不是同一次调用"而不相等,进 `repr` 则把库内部件泄进调用方的日志。 + + 洋葱各层经 `dataclasses.replace` 派生请求时保留**同一引用**(不是拷贝), + 重试/重问/分批才能共享同一个逻辑 ID 与计数。""" + @dataclass(frozen=True) class Usage: @@ -722,6 +821,8 @@ class OcrTextResult: latency_ms: int call_id: str raw: dict[str, Any] + call_stats: CallStats | None = None + """本次逻辑调用的统计快照(1.3.5);`None` = 未知。""" @dataclass(frozen=True) @@ -739,6 +840,8 @@ class OcrLayoutResult: latency_ms: int call_id: str raw: dict[str, Any] + call_stats: CallStats | None = None + """本次逻辑调用的统计快照(1.3.5);`None` = 未知。""" @dataclass(frozen=True) @@ -783,3 +886,5 @@ class EmbeddingResponse: call_id: str source_name: str cost: float | None = None + call_stats: CallStats | None = None + """本次逻辑调用(含全部分批)的统计快照(1.3.5);`None` = 未知。""" diff --git a/tests/unit/test_cache.py b/tests/unit/test_cache.py index 3f5b4b6..6e86044 100644 --- a/tests/unit/test_cache.py +++ b/tests/unit/test_cache.py @@ -818,3 +818,75 @@ class TestExplicitCacheMigration: finally: for transport in transports: await transport.aclose() + + +class TestCallStatsNotPoisoned: + """缓存不得回放历史统计(1.3.5 设计 §3)。 + + 统计描述**本次**调用;把上次那条存进去再放出来,等于对调用方谎称这次 + 重试了 N 次、耗了 M 毫秒。 + """ + + async def test_serialized_payload_carries_no_call_stats_key(self): + from polygateway.types import CallStats + + backend = InMemoryCache() + mw = _mw(backend) + stats = CallStats(logical_call_id="lc-1", attempts=3, total_latency_ms=900) + terminal = _Terminal(_resp(call_stats=stats)) + await mw(ChatRequest(messages=_MSGS), terminal) + key = build_cache_key("m", _MSGS, "proj", None) + stored = json.loads(await backend.get(key)) + assert "call_stats" not in stored # asdict 会把它摊成 dict,必须显式剔除 + + async def test_historic_dict_never_impersonates_call_stats(self): + """旧条目里的 `call_stats` dict 会被 `_RESPONSE_FIELDS` 放行,必须显式覆盖。 + + 不覆盖就会有一个 dict 冒充 `CallStats` 从公共 API 漏给调用方, + `resp.call_stats.attempts` 当场 `AttributeError`。 + """ + backend = InMemoryCache() + mw = _mw(backend) + key = build_cache_key("m", _MSGS, "proj", None) + poisoned = { + "content": "legacy", + "thinking": "", + "model": "m", + "provider": "p", + "prompt_tokens": 1, + "completion_tokens": 2, + "latency_ms": 30, + "ttft_ms": 5.0, + "max_inter_token_ms": 2.0, + "cache_hit": False, + "call_id": "orig", + "source_name": "s1", + "usage_source": "measured", + "call_stats": { + "logical_call_id": "stale-lc", + "attempts": 7, + "total_latency_ms": 9999, + }, + } + await backend.set(key, json.dumps(poisoned), ttl_s=100) + terminal = _Terminal(_resp()) + hit = await mw(ChatRequest(messages=_MSGS), terminal) + assert hit.content == "legacy" and terminal.calls == 0 # 真的走了缓存 + assert hit.call_stats is None # dict 不得冒充 CallStats + + async def test_cache_key_is_unchanged_by_the_new_field(self): + """新增内部字段不得扰动 key 公式,否则存量缓存全量冷启动(黄金值)。""" + from polygateway.types import _CallContext + + class _Clock: + def __call__(self): + return 1000.0 + + ctx = _CallContext(now=_Clock()) + bare = build_cache_key("m", _MSGS, "proj", None) + assert bare == build_cache_key("m", _MSGS, "proj", None) + # 带上下文的请求与不带的请求必须落在同一个 key 上 + with_ctx = ChatRequest(messages=_MSGS, call_context=ctx) + without = ChatRequest(messages=_MSGS) + assert with_ctx.cache_namespace == without.cache_namespace + assert digest_messages(with_ctx.messages) == digest_messages(without.messages) diff --git a/tests/unit/test_client.py b/tests/unit/test_client.py index 2f4d8e7..bc77f2c 100644 --- a/tests/unit/test_client.py +++ b/tests/unit/test_client.py @@ -1362,3 +1362,140 @@ async def test_synthetic_runtime_protocol_and_legacy_call_signatures(): assert client._transport._clients == {} finally: await client.aclose() + + +class _StatsClock: + """确定性单调钟;测试主动推进以断言"哪些区段计入了总耗时"。""" + + def __init__(self, start=1000.0): + self.t = start + + def __call__(self): + return self.t + + def advance(self, seconds): + self.t += seconds + + +class _TickingCache: + """假缓存后端: 每次 IO 推进注入钟。 + + 不推进时钟的替身会让"缓存 IO 计入总耗时"的断言退化成恒等于 0 的空转绿 + (计划 §T1 替身构造要求)。 + """ + + def __init__(self, clock, tick=0.25): + self._clock = clock + self._tick = tick + self._data = {} + + async def get(self, key): + self._clock.advance(self._tick) + return self._data.get(key) + + async def set(self, key, value, ttl_s): + self._clock.advance(self._tick) + self._data[key] = value + + +class _TickingRecorder: + """假 recorder: 写入时推进注入钟,用于断言内联遥测收尾计入总耗时。""" + + def __init__(self, clock, tick=0.5): + self._clock = clock + self._tick = tick + self.rows = [] + + async def record_llm_call(self, **fields): + self._clock.advance(self._tick) + self.rows.append(fields) + + +class TestLogicalCallStats: + """一次公开 chat 调用的统计(1.3.5 设计 §3)。""" + + _MSG = [{"role": "user", "content": "hi"}] + + async def test_success_reports_one_attempt(self): + async with _client() as client: + resp = await client.chat(self._MSG) + assert resp.call_stats is not None + assert resp.call_stats.attempts == 1 + assert resp.call_stats.logical_call_id + + async def test_concurrent_calls_do_not_share_counters_or_ids(self): + """同一 client 并发两路必须各自计数与各自 ID(库铁律「纯 asyncio 中立」)。 + + 上下文若被提升成 client 实例属性,这条就会红——那正是 VT + `evolve_llm = llm` 教训的同一形态。 + """ + async with _client() as client: + a, b = await asyncio.gather(client.chat(self._MSG), client.chat(self._MSG)) + assert a.call_stats.logical_call_id != b.call_stats.logical_call_id + assert a.call_stats.attempts == b.call_stats.attempts == 1 + + async def test_cache_hit_is_zero_attempts_with_a_fresh_logical_id(self): + """命中不产生网关调用 → 0 尝试;且是**新**逻辑调用,不回放历史统计。""" + clock = _StatsClock() + cache = _TickingCache(clock) + async with _client( + cache=cache, cache_namespace="proj", cache_ttl_s=600, now=clock + ) as client: + first = await client.chat(self._MSG) + second = await client.chat(self._MSG) + assert first.cache_hit is False and first.call_stats.attempts == 1 + assert second.cache_hit is True + assert second.call_stats.attempts == 0 + assert second.call_stats.logical_call_id != first.call_stats.logical_call_id + + async def test_cache_io_counts_into_total_latency(self): + """缓存读写是本次调用真实花掉的时间,必须进总耗时(设计 §3)。""" + clock = _StatsClock() + cache = _TickingCache(clock, tick=0.25) + async with _client( + cache=cache, cache_namespace="proj", cache_ttl_s=600, now=clock + ) as client: + hit = (await client.chat(self._MSG), await client.chat(self._MSG))[1] + # 命中路径只有一次 get(0.25s),无网关调用 + assert hit.call_stats.attempts == 0 + assert hit.call_stats.total_latency_ms == 250 + + async def test_inline_telemetry_teardown_counts_into_total_latency(self): + """成功响应的快照含返回前已完成的内联遥测耗时(设计 §6)。""" + clock = _StatsClock() + recorder = _TickingRecorder(clock, tick=0.5) + async with _client(telemetry=recorder, now=clock) as client: + resp = await client.chat(self._MSG) + assert recorder.rows # 确实写了行,否则本断言空转 + assert resp.call_stats.total_latency_ms == 500 + + async def test_milliseconds_not_seconds(self): + """毫秒/秒不混用: 1.5s 必须是 1500 而不是 1 或 1.5。""" + clock = _StatsClock() + recorder = _TickingRecorder(clock, tick=1.5) + async with _client(telemetry=recorder, now=clock) as client: + resp = await client.chat(self._MSG) + assert resp.call_stats.total_latency_ms == 1500 + + async def test_stats_work_without_any_telemetry(self): + """统计生效与否**不由 telemetry 是否启用决定**(设计 §3.5)。""" + async with _client(telemetry=None) as client: + resp = await client.chat(self._MSG) + assert resp.call_stats is not None and resp.call_stats.attempts == 1 + + async def test_failure_exception_carries_no_stats_attribute(self): + """本版**不向异常对象附加统计**(设计 §3.1): 第三方可能复用同一异常实例。""" + + def reject(request): + return httpx.Response(400, json={"error": {"message": "bad"}}) + + async with _client(handler=reject) as client: + with pytest.raises(RequestRejectedError) as exc: + await client.chat(self._MSG) + assert hasattr(exc.value, "call_stats") is False + + async def test_input_validation_stays_outside_the_stats_boundary(self): + """校验异常保持原行为,发生在统计边界之外(设计 §3)。""" + async with _client() as client: + with pytest.raises(ValueError, match="meta"): + await client.chat(self._MSG, meta={"BAD-KEY": 1}) diff --git a/tests/unit/test_embedding.py b/tests/unit/test_embedding.py index 94b2c6c..3384a75 100644 --- a/tests/unit/test_embedding.py +++ b/tests/unit/test_embedding.py @@ -600,3 +600,43 @@ class TestReasonlessTelemetryContract: assert seen == [{"model": "embed-1", "input": ["text"]}] finally: await transport.aclose() + + +class TestEmbedLogicalCallStats: + """分批共享同一逻辑调用(1.3.5 设计 §3/§3.5)。""" + + async def test_three_batches_count_three_attempts(self): + """分批是库的实现细节,但每批都真打了一次网关,故计 3 次尝试。""" + client, _ = _embed_client([_src()], ["ok", "ok", "ok"], batch_size=2) + resp = await client.embed(["a", "bb", "ccc", "dddd", "eeeee"]) + assert resp.call_stats is not None + assert resp.call_stats.attempts == 3 + + async def test_separate_calls_get_distinct_logical_ids(self): + """一次公开调用一个 ID: 两次 `embed` 不得共用同一个。 + + 共用就意味着上下文被提升成了 client 实例属性(库铁律禁止的形态)。 + """ + client, _ = _embed_client([_src()], ["ok"] * 5, batch_size=2) + first = await client.embed(["a", "bb", "ccc", "dddd", "eeeee"]) # 3 批 + second = await client.embed(["x", "y"]) # 1 批 + assert first.call_stats.attempts == 3 and second.call_stats.attempts == 1 + assert first.call_stats.logical_call_id != second.call_stats.logical_call_id + + async def test_retry_within_a_batch_is_counted(self): + client, _ = _embed_client([_src(), _src(name="e2")], [TransientError("t1"), "ok"]) + resp = await client.embed(["a"]) + assert resp.call_stats.attempts == 2 + + async def test_empty_input_is_zero_attempts_and_writes_no_telemetry_row(self): + """合法零尝试: 返回真实统计,且**不写任何遥测行**(设计 §3 M2)。 + + 与 cache_hit 不同——不要按"遥测必录"推断空输入也有台账行。 + """ + rec = _MemoryRecorder() + client, _ = _embed_client([_src()], [], telemetry=rec) + resp = await client.embed([]) + assert resp.call_stats is not None + assert resp.call_stats.attempts == 0 + assert resp.call_stats.logical_call_id # 真实 ID,不是空串 + assert rec.rows == [] # 零遥测行 diff --git a/tests/unit/test_ocr_client.py b/tests/unit/test_ocr_client.py index 00e435b..acac9c2 100644 --- a/tests/unit/test_ocr_client.py +++ b/tests/unit/test_ocr_client.py @@ -615,3 +615,34 @@ class TestReasonlessTelemetryContract: await getattr(client, method)(b"image") assert len(recorder.rows) == len(script) assert all(r["error"] and r["reasoning_effort"] is None for r in recorder.rows) + + +class TestOcrLogicalCallStats: + """OCR 两个公开方法各自拥有一次逻辑调用(1.3.5 设计 §3/§3.5)。""" + + async def test_text_success_counts_one_attempt(self): + client, _, _ = _client([_src()], ["text"]) + r = await client.recognize_text(b"jpg") + assert r.call_stats is not None and r.call_stats.attempts == 1 + + async def test_layout_two_http_calls_count_as_one_attempt(self): + """POST + ZIP GET 在同一次 transport 调用内,计 **1** 次尝试而非 2。 + + `attempts` 的语义是"调用 transport 端口的次数",不是 HTTP 请求条数。 + """ + client, _, _ = _client([_src()], ["layout"]) + r = await client.parse_layout(b"jpg") + assert r.call_stats is not None and r.call_stats.attempts == 1 + + async def test_retry_counts_every_attempt(self): + client, _, _ = _client([_src(), _src(name="m2")], [TransientError("t1"), "text"]) + r = await client.recognize_text(b"jpg") + assert r.call_stats.attempts == 2 + + async def test_input_validation_stays_outside_the_stats_boundary(self): + """`image` 类型/空校验先于上下文创建(M1 例外),保持原异常行为。""" + client, _, _ = _client([_src()], []) + with pytest.raises(TypeError): + await client.recognize_text("not-bytes") + with pytest.raises(ValueError): + await client.recognize_text(b"") diff --git a/tests/unit/test_retry.py b/tests/unit/test_retry.py index 96d566d..5a91b7a 100644 --- a/tests/unit/test_retry.py +++ b/tests/unit/test_retry.py @@ -5,6 +5,7 @@ """ import asyncio +import dataclasses import pytest @@ -776,3 +777,60 @@ class TestRateLimitPushback: await mw(_REQ) assert ei.value.reason == "retry_exhausted" assert len(transport.calls) == 3 + + +class TestLogicalAttemptCounting: + """尝试登记在 transport 调用**之前**(1.3.5 设计 §4)。 + + 登记点若挪到成功之后,失败与取消的尝试就会从计数里消失——而那正是 + 诊断时最需要看见的几次。 + """ + + def _ctx(self, clock): + from polygateway.types import _CallContext + + return _CallContext(now=clock) + + async def test_single_success_counts_one(self): + mw, _, _, _, _, clock = _harness([_src("a")], [_ok()]) + ctx = self._ctx(clock) + await mw(dataclasses.replace(_REQ, call_context=ctx)) + assert ctx.snapshot().attempts == 1 + + async def test_failed_retries_are_counted(self): + """两次可重试失败 + 一次成功 = 3 次尝试,不是 1 次。""" + mw, _, _, transport, _, clock = _harness( + [_src("a")], [TransientError("t1"), TransientError("t2"), _ok()] + ) + ctx = self._ctx(clock) + await mw(dataclasses.replace(_REQ, call_context=ctx)) + assert ctx.snapshot().attempts == 3 == len(transport.calls) + + async def test_budget_free_429_still_counts_as_an_attempt(self): + """429 免的是重试预算,不是"没发生过"——它确实打到了网关。""" + mw, _, _, transport, _, clock = _harness( + [_src("a")], + [ + TransientError("t1", status_code=429, retry_after_s=1.0), + TransientError("t2", status_code=429, retry_after_s=1.0), + _ok(), + ], + ) + ctx = self._ctx(clock) + await mw(dataclasses.replace(_REQ, call_context=ctx)) + assert ctx.snapshot().attempts == 3 == len(transport.calls) + + async def test_retry_exhausted_counts_every_attempt(self): + mw, _, _, transport, _, clock = _harness( + [_src("a")], [TransientError(str(i)) for i in range(5)], max_attempts=3 + ) + ctx = self._ctx(clock) + with pytest.raises(AllSourcesExhausted): + await mw(dataclasses.replace(_REQ, call_context=ctx)) + assert ctx.snapshot().attempts == 3 == len(transport.calls) + + async def test_absent_context_does_not_break_the_call(self): + """库内现场构造的 `ChatRequest` 没有上下文,不得因此报错(设计 §3.5)。""" + mw, _, _, _, _, _ = _harness([_src("a")], [_ok()]) + resp = await mw(_REQ) + assert resp.content == "ok" and _REQ.call_context is None diff --git a/tests/unit/test_types.py b/tests/unit/test_types.py index 6efafe8..4459383 100644 --- a/tests/unit/test_types.py +++ b/tests/unit/test_types.py @@ -610,3 +610,144 @@ class TestSourceConfigEffortNormalization: """非字符串同样只能是 `ValueError`: 公共入口不许把类型错误漏成 `AttributeError`。""" with pytest.raises(ValueError, match="推理档位"): _make_source(reasoning_effort=3) + + +class TestCallStatsAndContext: + """逻辑调用统计内核(1.3.5 设计 §3/§4)。""" + + def test_call_stats_is_frozen_snapshot(self): + from polygateway.types import CallStats + + stats = CallStats(logical_call_id="lc-1", attempts=2, total_latency_ms=15) + with pytest.raises(dataclasses.FrozenInstanceError): + stats.attempts = 3 + + def test_context_counts_attempts_and_freezes_elapsed(self): + """快照是同步冻结的时间切片: 登记两次尝试后耗时按注入钟折算成毫秒。""" + from polygateway.types import _CallContext + + clock = _FakeMonotonic() + ctx = _CallContext(now=clock) + clock.advance(1.5) + ctx.register_attempt() + ctx.register_attempt() + stats = ctx.snapshot() + assert stats.attempts == 2 + assert stats.total_latency_ms == 1500 # 秒→毫秒,不混用单位 + + def test_snapshot_is_repeatable_and_tracks_later_time(self): + from polygateway.types import _CallContext + + clock = _FakeMonotonic() + ctx = _CallContext(now=clock) + first = ctx.snapshot() + clock.advance(2.0) + second = ctx.snapshot() + assert first.total_latency_ms == 0 and second.total_latency_ms == 2000 + assert first.logical_call_id == second.logical_call_id + + def test_each_context_gets_its_own_logical_id(self): + from polygateway.types import _CallContext + + clock = _FakeMonotonic() + assert _CallContext(now=clock).logical_call_id != _CallContext(now=clock).logical_call_id + + def test_claim_terminal_is_true_once(self): + """终态去重位: 保证每逻辑调用至多写一条终态行(设计 §6 不变量 I3)。""" + from polygateway.types import _CallContext + + ctx = _CallContext(now=_FakeMonotonic()) + assert ctx.claim_terminal() is True + assert ctx.claim_terminal() is False + + def test_chat_request_context_does_not_affect_equality_or_repr(self): + """上下文是库内部件: 进 `compare`/`repr` 会污染既有请求语义与日志。""" + from polygateway.types import _CallContext + + ctx = _CallContext(now=_FakeMonotonic()) + bare = ChatRequest(messages=[{"role": "user", "content": "hi"}]) + with_ctx = dataclasses.replace(bare, call_context=ctx) + assert with_ctx.call_context is ctx + assert with_ctx == bare + assert "call_context" not in repr(with_ctx) + + def test_replace_preserves_the_same_context_reference(self): + """洋葱各层经 `replace` 派生请求,上下文必须是同一实例而非拷贝。""" + from polygateway.types import _CallContext + + ctx = _CallContext(now=_FakeMonotonic()) + req = ChatRequest(messages=[{"role": "user", "content": "hi"}], call_context=ctx) + derived = dataclasses.replace(req, stream=False) + assert derived.call_context is ctx + + def test_four_responses_default_call_stats_to_none(self): + """第三方合成响应的 `None` 表示未知,不得伪造 0(设计 §3)。""" + from polygateway.types import ( + EmbeddingResponse, + OcrLayoutResult, + OcrTextResult, + ) + + llm = LLMResponse( + content="c", + thinking="", + model="m", + provider="p", + prompt_tokens=1, + completion_tokens=1, + latency_ms=1, + ttft_ms=None, + max_inter_token_ms=None, + cache_hit=False, + call_id="c1", + ) + emb = EmbeddingResponse( + vectors=[], + dim=0, + model="m", + provider="p", + prompt_tokens=0, + usage_source="measured", + latency_ms=0, + call_id="c2", + source_name="s", + ) + text = OcrTextResult( + text="", source_name="s", usage=Usage(0, 0), latency_ms=0, call_id="c3", raw={} + ) + layout = OcrLayoutResult( + elements=[], + page_sizes=[], + source_name="s", + usage=Usage(0, 0), + latency_ms=0, + call_id="c4", + raw={}, + ) + assert (llm.call_stats, emb.call_stats, text.call_stats, layout.call_stats) == ( + None, + None, + None, + None, + ) + + def test_call_stats_is_exported_from_package_root(self): + """四份平铺字段会漂移,故统计以单一对象出现在公共 API(设计 §3)。""" + import polygateway + from polygateway.types import CallStats + + assert polygateway.CallStats is CallStats + assert "CallStats" in polygateway.__all__ + + +class _FakeMonotonic: + """确定性单调钟;不复用 contracts 的 FakeClock 以免 unit 反向依赖契约包。""" + + def __init__(self, start: float = 1000.0) -> None: + self.t = start + + def __call__(self) -> float: + return self.t + + def advance(self, seconds: float) -> None: + self.t += seconds From 393f2bf6170b4cc64c32bee9453213b16cdbb4d1 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 11:27:52 -0400 Subject: [PATCH 06/10] feat: record call observability columns and terminal failure rows Grow the telemetry contract from 26 to 36 fields and give every logical call a failure terminal row, so SQL can finally answer "how many calls failed" and "why did the whole pool die". Schema and port move together with the emitter writes in one commit: splitting them would ship columns that nothing populates. - schema: append 10 nullable columns (scope, operation, logical_call_id, event_kind, http_status_code, error_type, cause_type, error_body, attempts, total_latency_ms) to all five definition sites in one order - ports: 10 keyword-only parameters without defaults; the protocol signature is now the single source the assembly gate derives from - emitter: take domain exception objects instead of pre-flattened text and pin down the diagnostics in one helper; a relabelled 503 stays 503 and success rows leave all five columns NULL - emitter: reject recorders whose record_llm_call cannot accept the current field shape at assembly time, since _record would otherwise swallow the TypeError and drop every row while calls keep succeeding - clients: write at most one terminal row per logical call through a single shared exit, deduplicated by the call context; TelemetryMW stops writing terminals so the two sites cannot double count - clients: cancellation stays best effort and propagates, non-domain exceptions get no terminal row and keep their classification - transports: give _status_to_error an explicit operation and fix the historically mislabelled embedding HTTP failures - structured: promote the bounded error formatter so the reask feedback and the terminal explanation share one set of limits Terminal rows carry no cost and no tokens, so cost aggregation is unchanged; failure counts must now filter on event_kind. --- ...09-09-135-call-observability-validation.md | 141 ++++ src/polygateway/client.py | 27 +- src/polygateway/embedding.py | 90 ++- src/polygateway/middleware/retry.py | 15 +- src/polygateway/middleware/structured.py | 21 +- src/polygateway/middleware/telemetry.py | 317 ++++++-- src/polygateway/ocr.py | 149 +++- src/polygateway/ports.py | 21 +- src/polygateway/telemetry/schema.py | 62 +- src/polygateway/transports/openai_compat.py | 27 +- tests/unit/test_cache.py | 3 +- tests/unit/test_client.py | 148 +++- tests/unit/test_embedding.py | 7 +- tests/unit/test_ocr_client.py | 8 +- tests/unit/test_openai_compat.py | 3 +- tests/unit/test_ports.py | 53 +- tests/unit/test_pricing.py | 16 +- tests/unit/test_telemetry.py | 690 ++++++++++++++++-- tests/unit/test_usage_source_domain.py | 24 +- 19 files changed, 1628 insertions(+), 194 deletions(-) create mode 100644 research-wiki/findings/2026-09-09-135-call-observability-validation.md diff --git a/research-wiki/findings/2026-09-09-135-call-observability-validation.md b/research-wiki/findings/2026-09-09-135-call-observability-validation.md new file mode 100644 index 0000000..a0a3270 --- /dev/null +++ b/research-wiki/findings/2026-09-09-135-call-observability-validation.md @@ -0,0 +1,141 @@ +--- +type: finding +node_id: finding:2026-09-09-135-call-observability-validation +title: "1.3.5 T2/T3 验收证据:36 列遥测、失败终态与取消口径" +date: 2026-09-09 +--- + +# 1.3.5 T2/T3 验收证据 + +> 范围:**仅 T2(遥测 10 列 / 诊断保真 / scope+operation / 装配闸)与 T3(终态行 / 取消 / 统一出口)**, +> 外加计划 T4 中"签名与列数机械迁移"那一片(与 schema 同批完成,避免先提交 schema 却留写入缺键)。 +> **不含** T4 的 PG 集成、变异矩阵与文档同步——另任务承接,缺口见 §5。 +> 设计:`designs/2026-09-09-135-call-observability-design.md`;计划:`plans/2026-09-09-135-call-observability.md`。 +> 基线 HEAD `87c261b`(T1 已提交,1419 unit 全绿)。全部命令在 `PolyGateway` conda 环境执行。 + +## 1. 红绿证据链 + +TDD 纪律要求"先失败后通过",且红必须是**行为红**而非 import 红。下表每行都对应本会话的真实工具输出。 + +| # | 阶段 | 命令 | 结果 | +| --- | --- | --- | --- | +| 0 | 基线 | `pytest tests/unit -q` | **1419 passed**(对照底) | +| 1 | 红(T2/T3 目标行为) | `pytest tests/unit/test_telemetry.py -k "RowLevelObservability or RecorderShapeGate"` | **26 failed**,证据 `tests/outputs/135/red-01-emitter-rows.txt` | +| 2 | 实现后全量红面 | `pytest tests/unit -q` | **137 failed / 1308 passed**(机械迁移面暴露),`red-02-after-impl.txt` | +| 3 | 迁移中 | 同上 | 60 → 25 → 8 → 4 failed(`red-03`/`red-04`/`red-05`) | +| 4 | 绿 | `pytest tests/unit -q` | **1467 passed, 0 failed** | +| 5 | 静态门 | `make check`(ruff lint+format + import-linter) | **Contracts: 1 kept, 0 broken** | + +第 1 步的红是行为红而非 import 红:新用例调用的是**已存在**的 `TelemetryEmitter`, +失败形态是"缺 `scope=`/`operation=`/`stats=` 关键字"与"断言的列不存在",不是模块导不进来。 + +补测阶段另有两次真实红(均由 conda 实跑暴露、当场修正,非噪音): +`test_client.py` 的 `NameError: ResultInvalidError / ChatRequest`(漏 import)、 +`TestTerminalRowSqlSemantics` 的 `'coroutine' object has no attribute 'execute'`(async helper 漏 await)。 + +## 2. 核心要求逐条对应 + +### 2.1 36 列与 Emitter 完整一致(不留"schema 有列、写入缺键") + +一次性同批改完五处列定义 + 端口签名 + Emitter 写入,故不存在中间态。 + +| 判据 | 证据 | +| --- | --- | +| `len(COLUMNS) == 36`、物理列 37 | `test_telemetry.py::TestSchemaModule::test_columns_and_ddl_are_frozen`;`TestSQLiteSchemaMode` 断言 23 → 37 | +| 五处列序一致(DDL/BACKFILL×2/COLUMNS) | 新建库与 ALTER 追加列序同为 `_EXPECTED_COLUMNS` | +| 端口实测 36 字段、10 新参 keyword-only 且无默认值 | `test_ports.py::TestTelemetryRecorderSignature`(`inspect.signature` 实测,不凭记忆) | +| **Emitter 实参键集合 == `schema.COLUMNS`** | `TestEmitterRecorderContract` 三入口逐个断言 `set(rows[0]) == set(COLUMNS)` | +| 1.2.1 冻结 INSERT 未被改写 | 按 `_PRE_135_COLUMNS`(26 列)重现原文;全量 36 列另按占位符个数断言 | + +装配闸(C3):`_assert_recorder_shape` 在 `TelemetryEmitter.__init__` 做一次 `signature.bind`, +参数名**从协议签名派生**而非手抄第四份清单——`test_gate_derives_parameter_names_from_the_protocol` +用 monkeypatch 换掉协议后闸自动跟随,证明没有硬编码。旧签名 recorder / 无该方法 / 不可 inspect +一律装配期 `ValueError`(不是 warning:降级铁律管的是运行期写失败,不是配置错误)。 + +### 2.2 三个 client 的**真实链路**失败终态 + +不是只测 Emitter,而是驱动真实 client + MockTransport 到落库。 + +| 链路 | 用例 | 断言 | +| --- | --- | --- | +| chat 重试耗尽 | `TestChatTerminalFailureRows::test_retry_exhaustion_writes_exactly_one_terminal_row` | 3 条 attempt + **恰 1 条**终态;终态 `attempts == 3` | +| chat 结构化耗尽 | `test_structured_exhaustion_writes_the_only_failure_row` | attempt 行**全是成功行**,终态是唯一失败记录;error 含有界 `validation=`/`repair=` 且 < 1200 字符、不含 raw_text | +| chat 400 直拒 | `test_request_rejected_now_has_both_an_attempt_and_a_terminal_row` | 尝试行 + 终态行**各 1**(已批准的行数翻倍);两行共享同一 `logical_call_id` | +| embedding | `test_embedding.py::TestReasonlessTelemetryContract::test_failed_attempts_still_have_null_effort` | attempt 数 == 脚本长度,终态恰 1 | +| OCR 两方法 | `test_ocr_client.py` 同名用例(参数化 `recognize_text`/`parse_layout`) | 终态恰 1,且 `operation` 为**公开方法名** | + +去重:`test_terminal_row_is_written_once_per_logical_call` 连调出口 3 次,SQL 可见仍 1 条(`claim_terminal`)。 +非领域异常:`test_non_domain_exception_writes_no_terminal_row` —— `KeyError` 原样传播、**0 条**终态、分类不被改写。 + +**双写已消除**:`TelemetryMW` 的两个终态分支删除,改由三个 client 的公开边界经 +`emit_terminal_once` 统一写出;`TestTelemetryMW::test_scope_level_failure_is_not_written_here_anymore` +与 `test_cancellation_is_not_written_here_anymore` 锁死"本层不再写终态",防回归双计。 + +### 2.3 取消口径 + +- chat 取消:`test_cancellation_writes_at_most_one_terminal_row` —— 恰 1 条,`error == "cancelled"`, + 且 `error_type` 为 NULL(**字符串不解析猜诊断**)。 +- 三链路同策略"尽力写一条、允许 0";`emit_terminal_once` **不 shield、不开后台任务**, + 快照冻结是同步动作;写入 await 上再被取消则 `CancelledError` 原样传播(与 TelemetryMW 历史行为同款)。 + +### 2.4 SQL 不双计(按真实 SQLite 落库断言) + +`TestTerminalRowSqlSemantics` 用真实 `SQLiteRecorder` 驱动一次失败 chat 后直接查表: + +| 迁移影响 | 断言 | +| --- | --- | +| 失败计数判据 | `error IS NOT NULL` 得 **3**(2 尝试 + 1 终态),`event_kind='terminal_failure'` 得 **1** | +| 费用不双计 | 终态行 `cost IS NOT NULL` 计数为 **0**;且 `usage_source='unavailable'`、token 全 0 | +| 时延分组 | 终态 `latency_ms` ≥ 任何单次尝试(含退避),故看板必须按 `event_kind` 分组 | +| 双时钟微差 | 终态 `latency_ms == total_latency_ms`(同一份冻结快照) | +| 逻辑两列归属 | 尝试行 `attempts`/`total_latency_ms` 恒 NULL | +| §5 归因查询 | 同一 `logical_call_id` 同时给出整池 reason 文案与逐源 503 现场;终态 `http_status_code` 为 NULL(C1 不冒充) | + +### 2.5 诊断保真与 operation 修正 + +- 中转把 529 改写成 503 → **记 503 不猜回 529**;直接 529 记 529。 +- 空 `str()` 的 Connect/Read/Write/PoolTimeout → `cause_type` 落对应 httpx 类名,`error` 退回类名。 +- 成功行五列全 NULL(**不统一填 200**)。 +- `_status_to_error` 增 keyword `operation`:`embed()` 非 200 改传 `"embedding"`(修正历史误标), + 流式与非流式 chat 两处仍 `"chat"`(按实施计划 §2 归属表,未按设计行号误标)。 +- 新列 `operation` 恒为公开方法四值,**绝不读 `exc.operation`**: + `test_operation_is_given_by_the_call_site_not_the_exception` 用 `exc.operation="download_result"` 反证。 +- OCR `"类名: msg"` 前缀由出口的显式策略参数 `class_prefixed_error` 承载,不再三处各拼一遍。 + +## 3. 有界 validation 说明的单一所有者 + +`structured.py` 的 `_MAX_FEEDBACK_ERRORS`/`_MAX_ERROR_CHARS`/`_format_errors` 改名为 +公开的 `MAX_FEEDBACK_ERRORS`/`MAX_ERROR_CHARS`/`format_bounded_errors`, +由"重问反馈"与"终态结构化说明"两个消费者共同引用,**数值只有一份**。 +行为逐字不变(`test_structured.py` 22 项全绿,含反馈文案用例)。 + +## 4. 未改动确认(防越界) + +`errors.py`、`admission.py`、`ratelimit.py`、`breaker.py`、`sources.py`、`thinking.py`、 +`providers.py`、`telemetry/sqlite.py`、`telemetry/postgres.py`、`transports/monkey_ocr.py` +一字未动——两个 recorder 靠 `**fields` + `schema.COLUMNS` 自动吃到新列。 +缓存 key 公式、重试预算与退避、429 免预算、stall 算法、取消结算、推理能力表均未触碰。 +唯一顺带修正:`RetryMW.__init__` 的 `emitter` 注解由 `object | None` 收紧为 +`TelemetryEmitter | None`(TYPE_CHECKING 导入,同层不破分层契约),因本轮改了它的 `_emit`。 + +## 5. 明确缺口(交 T4 另任务,本轮不做) + +| 缺口 | 说明 | +| --- | --- | +| PG 存储兼容 | `tests/integration/test_postgres_telemetry.py` 的 `_EXPECTED_COLUMNS` 与字段字典**尚未补 10 列**;auto 追加 / manual 缺列裁剪 / 旧行 NULL / 新旧进程混写四项未跑(需真实 PG 沙箱) | +| 变异证据 | 计划 §4 的七项变异(计数位置、上下文复制、缓存回放、提前压平、终态双计费用、去 `claim_terminal`、闸改 warning)未执行 | +| 文档同步 | README「必录 26 字段」、`PGW_TELEMETRY_TEXT_CAP` 覆盖面、ARCHITECTURE §7.8、CHANGELOG、schemas/metrics 未更新 | +| 独立验证 | 未派全新上下文 verifier(合并前硬门) | +| slow / e2e | 未跑,属发布清单第 4 步 | + +## 6. 环境噪音记录(不是缺陷) + +自动检查器用**系统解释器 Python 3.13.9**(无 `redis`/`pydantic`/`httpx`/`loguru`/`asyncpg` 等依赖), +持续报 `test_client.py` 2 项失败与大量 "Import could not be resolved"、`StrEnum is unknown import symbol`。 +已核实为环境问题、非本轮引入: + +- 失败根因是 `ModuleNotFoundError: No module named 'redis'`(optional extra),本轮 diff 对 redis 零改动; +- 在**未改动的 HEAD** 上用同一系统解释器复跑,同样 2 failed / 90 passed; +- 项目强制环境 `conda run -n PolyGateway`(Python 3.12.13)下:`test_client.py` 98 passed、全量 1467 passed。 + +判据以 CLAUDE.md §2 规定的 conda 环境与 `make check` 为准。 diff --git a/src/polygateway/client.py b/src/polygateway/client.py index 9ac1a64..4534337 100644 --- a/src/polygateway/client.py +++ b/src/polygateway/client.py @@ -20,11 +20,12 @@ from polygateway.backends.memory.breaker import InMemoryGate from polygateway.backends.memory.cache import InMemoryCache from polygateway.backends.memory.limiter import InMemoryLimiter from polygateway.config import GatewaySettings +from polygateway.errors import PolyGatewayError from polygateway.middleware.base import compose from polygateway.middleware.cache import CacheMW from polygateway.middleware.retry import RetryMW from polygateway.middleware.structured import StructuredMW -from polygateway.middleware.telemetry import TelemetryEmitter, TelemetryMW +from polygateway.middleware.telemetry import TelemetryEmitter, TelemetryMW, emit_terminal_once from polygateway.ports import TelemetryStatusProvider from polygateway.pricing import PricingTable from polygateway.providers import get_provider @@ -239,7 +240,7 @@ class GatewayClient: rng: Any = random.random, ) -> None: emitter = ( - TelemetryEmitter(telemetry, pricing=pricing, text_cap=text_cap) + TelemetryEmitter(telemetry, scope=scope, pricing=pricing, text_cap=text_cap) if telemetry is not None else None ) @@ -393,7 +394,27 @@ class GatewayClient: meta=dimensions, call_context=context, ) - response = await self._handler(request) + try: + response = await self._handler(request) + except PolyGatewayError as exc: + # 统计边界内的一切领域失败均尝试写一条终态行(1.3.5 设计 §6 I3), + # 包括已有 attempt 错误行的 RequestRejected / ResultInvalid——两类行描述 + # 的不是同一件事(尝试 vs 逻辑终态),由 `event_kind` 区分 + await emit_terminal_once( + self._emitter, request=request, context=context, error=exc, operation="chat" + ) + raise + except asyncio.CancelledError: + # 尽力而为且**取消优先**: 不 shield、不开后台任务;写入那一次 await 上 + # 再被取消则 `CancelledError` 照常传播(与 TelemetryMW 历史行为同款) + await emit_terminal_once( + self._emitter, + request=request, + context=context, + error="cancelled", + operation="chat", + ) + raise # 快照在返回前冻结: 故它含缓存命中路径与已完成的内联遥测耗时 return dataclasses.replace(response, call_stats=context.snapshot()) diff --git a/src/polygateway/embedding.py b/src/polygateway/embedding.py index 44ac701..7b3a728 100644 --- a/src/polygateway/embedding.py +++ b/src/polygateway/embedding.py @@ -42,7 +42,7 @@ from polygateway.middleware.admission import SourceAdmission, settle_and_release from polygateway.middleware.breaker import BreakerGate from polygateway.middleware.ratelimit import QuotaGate from polygateway.middleware.retry import StallClock, _failure_reason, backoff_delay -from polygateway.middleware.telemetry import TelemetryEmitter +from polygateway.middleware.telemetry import TelemetryEmitter, emit_terminal_once from polygateway.types import ( ChatRequest, EmbeddingResponse, @@ -129,7 +129,9 @@ class EmbeddingClient: self._transport = transport self._retry = retry self._emitter = ( - TelemetryEmitter(telemetry, pricing=pricing, text_cap=text_cap) if telemetry else None + TelemetryEmitter(telemetry, scope=self._scope, pricing=pricing, text_cap=text_cap) + if telemetry + else None ) self._telemetry = telemetry # 限流/熔断后端在此之外只以 QuotaGate/BreakerGate 的形态存在,自持一份 @@ -203,6 +205,38 @@ class EmbeddingClient: source_name="", call_stats=context.snapshot(), ) + try: + return await self._embed_all( + texts, session_id, parent_call_id, dimension_tenant_id, dimensions, context + ) + except PolyGatewayError as exc: + await self._emit_terminal( + texts, session_id, parent_call_id, dimension_tenant_id, dimensions, context, exc + ) + raise + except asyncio.CancelledError: + # 三条链路同一口径尽力写一条(允许 0 条);取消优先,不 shield + await self._emit_terminal( + texts, + session_id, + parent_call_id, + dimension_tenant_id, + dimensions, + context, + "cancelled", + ) + raise + + async def _embed_all( + self, + texts: list[str], + session_id: str | None, + parent_call_id: str | None, + tenant_id: str | None, + meta: dict[str, Any], + context: _CallContext, + ) -> EmbeddingResponse: + """切批串行执行并合并;无源的 raise 必须在本方法内——否则无源终态行写不出。""" if not self._sources: raise AllSourcesExhausted(scope=self._scope, reason="no_sources", retry_after_s=0.0) outcomes = [] @@ -212,14 +246,50 @@ class EmbeddingClient: texts[start : start + self._batch_size], session_id, parent_call_id, - dimension_tenant_id, - dimensions, + tenant_id, + meta, context, ) ) # 全批共享同一上下文,故分批是实现细节而非 N 次独立逻辑调用 return dataclasses.replace(self._merge(outcomes), call_stats=context.snapshot()) + async def _emit_terminal( + self, + texts: list[str], + session_id: str | None, + parent_call_id: str | None, + tenant_id: str | None, + meta: dict[str, Any], + context: _CallContext, + error: PolyGatewayError | str, + ) -> None: + """终态行的请求摘要(设计 §6 M4): 计数占位 + 第一批截断文本。 + + 描述的是**本次调用的整体输入**但不扩大单行正文预算: 失败批的具体文本由同 + `logical_call_id` 的 attempt 行给出,终态行不保存全量原输入。 + """ + batches = math.ceil(len(texts) / self._batch_size) + messages = [{"role": "user", "content": f""}] + # 与逐批行同款构造(至多 `batch_size` 条、每条 200 字符) + messages += [ + {"role": "user", "content": t[:_TELEMETRY_TEXT_CAP]} for t in texts[: self._batch_size] + ] + await emit_terminal_once( + self._emitter, + request=ChatRequest( + messages=messages, + session_id=session_id, + parent_call_id=parent_call_id, + tenant_id=tenant_id, + meta=meta, + call_context=context, + ), + context=context, + error=error, + operation="embed", + ) + # —— 治理循环(与 RetryMW 同构;设计 §7.1 已声明的有限重复)—— async def _embed_batch( @@ -300,6 +370,7 @@ class EmbeddingClient: parent_call_id, tenant_id, meta, + context, result, ) return _BatchOutcome(result, source, call_id, latency_ms) @@ -314,6 +385,7 @@ class EmbeddingClient: parent_call_id, tenant_id, meta, + context, error=exc, ) raise @@ -329,6 +401,7 @@ class EmbeddingClient: parent_call_id, tenant_id, meta, + context, error="cancelled", ) raise @@ -349,6 +422,7 @@ class EmbeddingClient: parent_call_id, tenant_id, meta, + context, error=exc, ) return _FailedBatch(exc, immediate=dead) @@ -385,8 +459,9 @@ class EmbeddingClient: parent_call_id: str | None, tenant_id: str | None, meta: dict[str, Any], + context: _CallContext, result: EmbeddingTransportResult | None = None, - error: object | None = None, + error: PolyGatewayError | str | None = None, ) -> None: """逐批遥测(经同一 Emitter): messages=截断 texts、向量绝不入库。""" if self._emitter is None: @@ -399,6 +474,7 @@ class EmbeddingClient: parent_call_id=parent_call_id, tenant_id=tenant_id, meta=meta, + call_context=context, ) response = None if result is not None: @@ -423,10 +499,12 @@ class EmbeddingClient: call_id=call_id, latency_ms=int((self._now() - started) * 1000), response=response, - error=None if error is None else str(error), + # 异常对象原样下传: 状态码/底层异常类型/网关正文在 Emitter 内定型 + error=error, # embedding payload 硬编码 {model, input},从不带推理参数;源上即便 # 误配了 ENABLE_THINKING,记一个档也是替这次调用声称它没做过的事 reasoning_applies=False, + operation="embed", ) def _merge(self, outcomes: list[_BatchOutcome]) -> EmbeddingResponse: diff --git a/src/polygateway/middleware/retry.py b/src/polygateway/middleware/retry.py index 0a3a1ea..7f66cd5 100644 --- a/src/polygateway/middleware/retry.py +++ b/src/polygateway/middleware/retry.py @@ -42,6 +42,7 @@ from polygateway.types import LLMResponse if TYPE_CHECKING: from collections.abc import AsyncIterator, Awaitable, Callable + from polygateway.middleware.telemetry import TelemetryEmitter from polygateway.ports import ( GateDecision, Permit, @@ -181,7 +182,7 @@ class RetryMW: circuit_open: str = "fail_fast", cooldown_memo: SourceCooldownMemo | None = None, pacer: AdaptivePacer | None = None, - emitter: object | None = None, + emitter: TelemetryEmitter | None = None, now: Callable[[], float] = time.monotonic, sleep: Callable[[float], Awaitable[None]] = asyncio.sleep, rng: Callable[[], float] = random.random, @@ -413,9 +414,13 @@ class RetryMW: started: float, *, response: LLMResponse | None = None, - error: object | None = None, + error: PolyGatewayError | str | None = None, ) -> None: - """逐次遥测(经注入的单一 Emitter);遥测失败不得影响调用(铁律)。""" + """逐次遥测(经注入的单一 Emitter);遥测失败不得影响调用(铁律)。 + + 异常**对象原样下传**而非先 `str()` 压平(1.3.5 设计 §5): 状态码、底层异常 + 类型与网关响应体已经在异常上了,在这里压平就是把它们丢掉。 + """ if self._emitter is None: return try: @@ -425,9 +430,11 @@ class RetryMW: call_id=call_id, latency_ms=int((self._now() - started) * 1000), response=response, - error=None if error is None else str(error), + error=error, # chat 路径是唯一带推理参数的路径,故实发档由这里的响应说了算 reasoning_applies=True, + # 公开方法四值之一;本中间件只服务 chat 洋葱 + operation="chat", ) except asyncio.CancelledError: raise diff --git a/src/polygateway/middleware/structured.py b/src/polygateway/middleware/structured.py index 2a70b63..6887676 100644 --- a/src/polygateway/middleware/structured.py +++ b/src/polygateway/middleware/structured.py @@ -14,6 +14,8 @@ from typing import TYPE_CHECKING from polygateway.errors import ResultInvalidError if TYPE_CHECKING: + from collections.abc import Sequence + from polygateway.ports import CallNext, StructuredOutputStrategy from polygateway.types import ChatRequest, LLMResponse @@ -22,12 +24,18 @@ _FEEDBACK_TEMPLATE = ( "Your previous reply was not valid JSON matching the required schema. " "Errors: {errors}. Reply with ONLY the corrected JSON object." ) -_MAX_FEEDBACK_ERRORS = 3 -_MAX_ERROR_CHARS = 200 +MAX_FEEDBACK_ERRORS = 3 +MAX_ERROR_CHARS = 200 -def _format_errors(errors: list[str]) -> str: - clipped = [e[:_MAX_ERROR_CHARS] for e in errors[:_MAX_FEEDBACK_ERRORS]] +def format_bounded_errors(errors: Sequence[str]) -> str: + """校验错误的有界拼装: 至多 3 条 × 每条 200 字符。 + + **本模块是这条规则的所有者**: 重问反馈文案与 1.3.5 终态行的结构化说明 + 两个消费者共用同一份实现与同一组数值——数值复制成两份必然漂移,而漂移后 + "模型看到的错误"与"台账里记的错误"就不再是同一件事。行为与重命名前逐字相同。 + """ + clipped = [e[:MAX_ERROR_CHARS] for e in list(errors)[:MAX_FEEDBACK_ERRORS]] return "; ".join(clipped) if clipped else "output could not be parsed" @@ -104,7 +112,10 @@ class StructuredMW: messages = [ *current.messages, {"role": "assistant", "content": bad_content}, - {"role": "user", "content": _FEEDBACK_TEMPLATE.format(errors=_format_errors(errors))}, + { + "role": "user", + "content": _FEEDBACK_TEMPLATE.format(errors=format_bounded_errors(errors)), + }, ] reask = dataclasses.replace(current, messages=messages) if self._escalation is not None: diff --git a/src/polygateway/middleware/telemetry.py b/src/polygateway/middleware/telemetry.py index 3b4c0f6..d728313 100644 --- a/src/polygateway/middleware/telemetry.py +++ b/src/polygateway/middleware/telemetry.py @@ -2,13 +2,19 @@ Emitter 是全库**唯一**调用 `record_llm_call` 的地方(三项目 4 处逐字复制 15 参调用的教训)。分工: RetryMW 经 Emitter 逐次记录每次尝试;TelemetryMW -(最外层)只记尝试层看不见的事件——缓存命中、scope 级失败、取消; -RequestRejected/ResultInvalid 已被尝试层记录,最外层放行不重复记。 +(最外层)只记尝试层看不见的缓存命中;而**终态失败行**由三个 client 的公开 +边界经 `emit_terminal_once` 统一写出(1.3.5)——两处同时写就会双计。 + +一行遥测属于三类事件之一(`event_kind`): `attempt`(一次尝试)、`cache_hit` +(未产生网关调用)、`terminal_failure`(一次**逻辑调用**的失败终态)。后两者与 +前者**不是重复事实**,故统计失败调用次数只能取 `terminal_failure`, +不得按 `error IS NOT NULL` 跨两类直接计数(设计 §6/§8)。 """ from __future__ import annotations import asyncio +import inspect import json import time import uuid @@ -17,12 +23,10 @@ from typing import TYPE_CHECKING from loguru import logger -from polygateway.errors import ( - GatewayUnavailableError, - GovernanceBackendError, - SourceNotConfiguredError, -) +from polygateway.errors import PolyGatewayError, ResultInvalidError from polygateway.middleware.cache import digest_messages +from polygateway.middleware.structured import MAX_ERROR_CHARS, format_bounded_errors +from polygateway.ports import TelemetryRecorder from polygateway.thinking import effective_effort from polygateway.types import Effort, ThinkingObservation, canonical_sampling_json, merge_sampling @@ -30,9 +34,17 @@ if TYPE_CHECKING: from collections.abc import Callable, Mapping from typing import Any - from polygateway.ports import CallNext, TelemetryRecorder + from polygateway.ports import CallNext from polygateway.pricing import PricingTable - from polygateway.types import ChatRequest, LLMResponse, SourceConfig + from polygateway.types import ( + CallOperation, + CallStats, + ChatRequest, + EventKind, + LLMResponse, + SourceConfig, + _CallContext, + ) def _canonical_meta_json(meta: Mapping[str, Any]) -> str: @@ -225,26 +237,159 @@ class _AttemptUsage: ) +@dataclass(frozen=True) +class _ErrorFields: + """一行遥测的错误列;未知一律 `None`。 + + 存在的理由是把"三种入参形态 × 两类行"的定型规则收敛到**一处**: + 改前调用方先 `str(exc)` 压平,状态码、底层异常类型与网关正文全部丢失。 + """ + + error: str | None = None + error_type: str | None = None + cause_type: str | None = None + http_status_code: int | None = None + error_body: str | None = None + + +def _structured_detail(exc: ResultInvalidError) -> str: + """结构化阶梯耗尽的**有界**说明(设计 §5 C2)。 + + `ResultInvalidError("结构化输出阶梯耗尽")` 的 message 不含校验与修复错误,而该 + 失败发生在 StructuredMW 之上——RetryMW 侧的 attempt 行全是**成功行**,终态行是 + 唯一记录。故把说明并入现有 `error` 串。 + + **不含 `raw_text`**: 它是模型正文,attempt 行的 `response` 列已按 `text_cap` 记过 + 一份;再存一份等于绕过既有的正文预算。条数与限长复用 `structured.py` 的同一 + 套常量(重问反馈与本说明同一口径),数值只有一份。 + """ + parts: list[str] = [] + if exc.repair_error: + parts.append(f"repair={exc.repair_error[:MAX_ERROR_CHARS]}") + if exc.validation_errors: + parts.append(f"validation={format_bounded_errors(exc.validation_errors)}") + return " | ".join(parts) + + +def _error_fields( + error: PolyGatewayError | str | None, + *, + event_kind: EventKind, + class_prefixed: bool, +) -> _ErrorFields: + """三种入参形态的唯一定型点(设计 §5/§6)。 + + - `None` → 全 None(成功行不统一填 200: 那会让"有状态码"不再等价于"失败了")。 + - `str`(取消路径的 `"cancelled"`)→ 原样落 `error`,**不解析字符串猜诊断**。 + - 领域异常 → 只读它既有的属性,不遍历任意对象、不猜正文。 + + **终态行的三列恒为 NULL(C1 红线)**: `GatewayUnavailableError` 家族从不携带 + 状态码与响应体,NULL 正是它自身的真实状态——把最后一次 attempt 的状态码与正文 + 搬上来,就是拿最后一个源冒充整池归因。逐源现场由同一 `logical_call_id` 的 + attempt 行给出。 + """ + if error is None: + return _ErrorFields() + if isinstance(error, str): + return _ErrorFields(error=error) + name = type(error).__name__ + # 空 `str()` 退回类名(httpx 的 Connect/Read/Write/PoolTimeout 文案就是空的); + # `class_prefixed` 是 OCR 的既有口径(按类名归组的 metric),故逐字保留它的拼法 + text = f"{name}: {error}" if class_prefixed else (str(error) or name) + if event_kind == "terminal_failure": + if isinstance(error, ResultInvalidError): + detail = _structured_detail(error) + if detail: + text = f"{text} | {detail}" + return _ErrorFields(error=text, error_type=name) + cause = error.__cause__ + return _ErrorFields( + error=text, + error_type=name, + cause_type=type(cause).__name__ if cause is not None else None, + # getattr 而非直读: 本函数在 `_record` 的降级 try **之外**求值, + # 一个非领域异常误传进来不得把一次真实失败换成 AttributeError + http_status_code=getattr(error, "status_code", None), + # 空串归 None: 既有 `body_text` 的缺省就是空串,而本列的语义是"未知" + error_body=getattr(error, "body_text", "") or None, + ) + + +def _assert_recorder_shape(recorder: TelemetryRecorder) -> None: + """装配期一次 `signature.bind` 形状校验: 不执行写入,只证明该形状能被接受。 + + `_record` 的 `except Exception` 会把旧 recorder 的 `TypeError` 吞成 warning, + 后果是自定义 recorder 在下游升级后**100% 丢遥测且调用照常成功**——正是 + "遥测必录"要防的形态,而文档级迁移清单挡不住它。故在装配期当场报错 + (不是 warning: 降级方向的铁律管的是**运行期写失败**,不是装配错误)。 + + 参数名从 `TelemetryRecorder.record_llm_call` 的协议签名**派生**(不手抄第四份 + 字段清单),绑定用哨兵 `None`,不读任何真实请求数据;`**kwargs` + (VAR_KEYWORD)自动通过。不可 inspect(C 实现等)同样按配置错误报错——宁可 + 装配不起来,不进入"运行期静默丢行"。 + + 边界诚实声明: 它只证明该形状能被接受,**不能证明函数体真的落这些列**。 + + Raises: + ValueError: 签名不符、不可 inspect,或协议本身不可 inspect。 + """ + try: + # 模块全局查找而非常量快照: 协议改了,闸就跟着改(测试可据此机械验证) + protocol = inspect.signature(TelemetryRecorder.record_llm_call).parameters + except (TypeError, ValueError) as exc: # pragma: no cover - 协议一向可 inspect + raise ValueError(f"TelemetryRecorder.record_llm_call 签名不可读取: {exc}") from exc + sentinels = {name: None for name in protocol if name != "self"} + label = type(recorder).__name__ + method = getattr(recorder, "record_llm_call", None) + if method is None: + # 连方法都没有: 比旧签名更明确的配置错误。不让它以裸 AttributeError + # 逆流而上——那不属错误四分类,且现场离"注错了东西"这个真因很远 + raise ValueError( + f"注入的遥测 recorder {label} 没有 record_llm_call 方法,不满足 TelemetryRecorder 端口" + ) + try: + signature = inspect.signature(method) + except (TypeError, ValueError) as exc: + raise ValueError( + f"遥测 recorder {label} 的 record_llm_call 不可 inspect(如 C 实现)," + "无法在装配期确认它接受当前字段形状;请换成 Python 实现或包一层" + ) from exc + try: + signature.bind(**sentinels) + except TypeError as exc: + raise ValueError( + f"遥测 recorder {label} 的 record_llm_call 签名与 TelemetryRecorder 不符" + f"(当前 {len(sentinels)} 个字段): {exc}。" + "这一条故意在装配期报错——放行的后果是每行遥测都被降级成 warning 后丢弃" + ) from exc + + class TelemetryEmitter: - """从请求与结果组装 26 字段并写入 recorder;一切写失败降级 warning。""" + """从请求与结果组装 36 字段并写入 recorder;一切写失败降级 warning。""" def __init__( self, recorder: TelemetryRecorder, *, + scope: str, pricing: PricingTable | None = None, text_cap: int | None, ) -> None: - """`text_cap` 无默认值是有意的: 它是关键行为参数,漏传即静默改变落库正文。 + """`text_cap` 与 `scope` 无默认值是有意的: 两者都是关键行为参数。 + `text_cap` 漏传即静默改变落库正文;`scope` 漏传则三类行都失去池名 + ——终态失败可能根本没选出源,但 scope 始终已知,不拿 `source_name` 顶替。 本类是库内部类,唯一构造者是三个公共 Client,必填能保证没有一处漏传。 - 同理,值域校验也放在这一处: 三个 Client 的 `text_cap` 全部汇流到这里, + + 同理,值域校验与**装配闸**都放在这一处: 三个 Client 全部汇流到这里, `GatewaySettings` 那道只管 env 一条路,而直接构造 Client 是库承诺的另一 - 条公共装配路——`text_cap=0` 会让每条正文只剩一个省略标记(P5 不得静默)。 + 条公共装配路(`text_cap=0` 会让每条正文只剩一个省略标记;P5 不得静默)。 """ if text_cap is not None and text_cap <= 0: raise ValueError(f"text_cap 必须 > 0(不截断请传 None): {text_cap}") + _assert_recorder_shape(recorder) self._recorder = recorder + self._scope = scope self._pricing = pricing self._text_cap = text_cap @@ -256,8 +401,10 @@ class TelemetryEmitter: call_id: str, latency_ms: int, response: LLMResponse | None, - error: str | None, + error: PolyGatewayError | str | None, reasoning_applies: bool, + operation: CallOperation, + class_prefixed_error: bool = False, ) -> None: """逐次尝试记录(三个 Client 的重试层调用);失败尝试无用量可言,记 0 并标 unavailable。 @@ -266,7 +413,11 @@ class TelemetryEmitter: 共用同一个 `SourceConfig` 类型,一个误配了 `ENABLE_THINKING` 的 embedding 源 会让下面的回落算出 `auto`,给一次从来不带推理参数的调用挂上一个从未发出过的 档。**不设默认值**: 与 `TelemetryRecorder` 同一约定,库外无第三方调用者,漏传 - 当场 TypeError,好过被静默当成"没表态"。 + 当场 TypeError,好过被静默当成"没表态"。`operation` 同理且另有一层: + 它只能由调用点给定,**绝不读 `exc.operation`**(后者是 HTTP 子操作)。 + + `error` 收**领域异常对象**而非预先 `str()` 压平的文本: 状态码/底层异常类型/ + 网关正文在此提取成四列(设计 §5)。取消路径仍传既有字符串 `"cancelled"`。 """ usage = _AttemptUsage.of(response) await self._record( @@ -284,7 +435,7 @@ class TelemetryEmitter: ttft_ms=usage.ttft_ms, max_inter_token_ms=usage.max_inter_token_ms, cache_hit=False, - error=error, + errors=_error_fields(error, event_kind="attempt", class_prefixed=class_prefixed_error), cached_prompt_tokens=usage.cached_prompt_tokens, model_reported=usage.model_reported, reasoning_tokens=usage.reasoning_tokens, @@ -296,10 +447,19 @@ class TelemetryEmitter: reasoning_effort=_attempt_effort( request=request, source=source, response=response, applies=reasoning_applies ), + operation=operation, + event_kind="attempt", + attempts=None, + total_latency_ms=None, ) - async def emit_cache_hit(self, *, request: ChatRequest, response: LLMResponse) -> None: - """缓存命中记录: cache_hit=True、latency_ms=0(VT 同款)。""" + async def emit_cache_hit( + self, *, request: ChatRequest, response: LLMResponse, operation: CallOperation + ) -> None: + """缓存命中记录: cache_hit=True、latency_ms=0(VT 同款)。 + + 逻辑计数两列恒 NULL: 本行描述的是"一次命中",不是一次逻辑调用的终态。 + """ await self._record( request=request, call_id=response.call_id, @@ -315,7 +475,7 @@ class TelemetryEmitter: ttft_ms=None, max_inter_token_ms=None, cache_hit=True, - error=None, + errors=_ErrorFields(), # 决策 B1: 与 model/prompt_tokens 同一口径,原样回放历史值。 # 统计供应商缓存命中率必须带 WHERE cache_hit = false,否则重复计数。 cached_prompt_tokens=response.cached_prompt_tokens, @@ -334,12 +494,28 @@ class TelemetryEmitter: # 与 sampling 同一口径: 命中行没有选中源,源级档位与 `nearest` 映射 # 都无从谈起,只记调用方这次要的档(response 里那个是历史那次实发的) reasoning_effort=_normalize_effort(request.reasoning_effort), + operation=operation, + event_kind="cache_hit", + attempts=None, + total_latency_ms=None, ) async def emit_terminal_failure( - self, *, request: ChatRequest, call_id: str, latency_ms: int, error: str + self, + *, + request: ChatRequest, + call_id: str, + error: PolyGatewayError | str, + operation: CallOperation, + stats: CallStats, + class_prefixed_error: bool = False, ) -> None: - """scope 级失败/取消记录: 无具体源,溯源字段置空标记。""" + """一次**逻辑调用**的失败终态: 无具体源,溯源字段置空标记。 + + `latency_ms` 与 `total_latency_ms` 同取**同一份冻结快照**,避免双时钟微差; + 故本方法不再收 `latency_ms`。token 与 cost 一律不从 attempt 行复制 + (费用聚合仍只由 attempt / cache_hit 行决定,口径不变)。 + """ await self._record( request=request, call_id=call_id, @@ -351,11 +527,13 @@ class TelemetryEmitter: prompt_tokens=0, completion_tokens=0, usage_source="unavailable", - latency_ms=latency_ms, + latency_ms=stats.total_latency_ms, ttft_ms=None, max_inter_token_ms=None, cache_hit=False, - error=error, + errors=_error_fields( + error, event_kind="terminal_failure", class_prefixed=class_prefixed_error + ), cached_prompt_tokens=None, model_reported=None, reasoning_tokens=None, @@ -368,6 +546,10 @@ class TelemetryEmitter: meta=request.meta, # 可能根本没选出源,故与 sampling 同样只取请求档 reasoning_effort=_normalize_effort(request.reasoning_effort), + operation=operation, + event_kind="terminal_failure", + attempts=stats.attempts, + total_latency_ms=stats.total_latency_ms, ) async def _record( @@ -387,7 +569,9 @@ class TelemetryEmitter: ttft_ms: float | None, max_inter_token_ms: float | None, cache_hit: bool, - error: str | None, + # 1.3.5: 错误四列已由 `_error_fields` 定型(三种入参形态 × 两类行的唯一规则所有者), + # 本方法只搬运——拆成五个平铺参数就是把"一处定型"换回"三处各自拼" + errors: _ErrorFields, cached_prompt_tokens: int | None, model_reported: str | None, sampling: str | None, @@ -403,6 +587,11 @@ class TelemetryEmitter: # 注释),本方法只搬运——把定型放这里就得再传一遍 response/source,等于把 # "唯一 record_llm_call 调用点"换成"两处口径判断",那正是要避免的复制 reasoning_effort: str | None, + # —— 1.3.5: 行形态与逻辑调用快照 —— + operation: CallOperation, + event_kind: EventKind, + attempts: int | None, + total_latency_ms: int | None, ) -> None: try: # 成本换算(M2 §6): 成功行按单价换算;缓存命中 0.0(未产生新调用); @@ -413,7 +602,7 @@ class TelemetryEmitter: # 用量不可得: 宁可算不出成本,也不算错成本(解耦设计 §3.1 不变式)。 # 必须排在 cache_hit 之后——缓存命中未产生新调用,0.0 是事实而非未知 cost = None - elif error is None and model and self._pricing is not None: + elif errors.error is None and model and self._pricing is not None: cost = self._pricing.cost( model, prompt_tokens, completion_tokens, cached_prompt_tokens ) @@ -425,6 +614,7 @@ class TelemetryEmitter: _cap_messages(digest_messages(request.messages), self._text_cap), ensure_ascii=False, ) + context = request.call_context await self._recorder.record_llm_call( call_id=call_id, parent_call_id=request.parent_call_id, @@ -442,7 +632,7 @@ class TelemetryEmitter: ttft_ms=ttft_ms, max_inter_token_ms=max_inter_token_ms, cache_hit=cache_hit, - error=error, + error=errors.error, cost=cost, cached_prompt_tokens=cached_prompt_tokens, model_reported=model_reported, @@ -457,6 +647,19 @@ class TelemetryEmitter: # Postgres 那一路悄悄少一列数据 thinking_observation=_normalize_observation(thinking_observation), reasoning_effort=reasoning_effort, + # —— 1.3.5 十列 —— + scope=self._scope, + # 调用点给定的公开方法四值,**绝不读 `exc.operation`**(设计 §5 I1/I2) + operation=operation, + # 上下文缺席(库内现场构造的 ChatRequest)→ NULL,**不造 ID**(I5) + logical_call_id=None if context is None else context.logical_call_id, + event_kind=event_kind, + http_status_code=errors.http_status_code, + error_type=errors.error_type, + cause_type=errors.cause_type, + error_body=errors.error_body, + attempts=attempts, + total_latency_ms=total_latency_ms, ) except asyncio.CancelledError: raise @@ -464,8 +667,45 @@ class TelemetryEmitter: logger.warning("遥测记录失败(降级不冒泡): {}", exc) +async def emit_terminal_once( + emitter: TelemetryEmitter | None, + *, + request: ChatRequest, + context: _CallContext, + error: PolyGatewayError | str, + operation: CallOperation, + class_prefixed_error: bool = False, +) -> None: + """三个 client 共用的**终态唯一出口**: 去重 + 同步冻结快照 + best effort 写入。 + + 去重由 `claim_terminal()` 承担(每逻辑调用至多一条终态行);`emitter is None` + 或已写过 → 直接返回。写入侧异常按既有降级只落 warning(在 `_record` 内)。 + + **`CancelledError` 原样传播**(取消优先,不 shield、不开后台任务): 这一次 + `await` 本身就是新的取消点,外部取消落在它上时调用方会看到 `CancelledError` + 而非领域错误——与 TelemetryMW 的历史行为同款,已经人类批准(设计 §6/§10)。 + 快照冻结是**同步**动作,故终态行不含它自身的写入耗时。 + """ + if emitter is None or not context.claim_terminal(): + return + stats = context.snapshot() + await emitter.emit_terminal_failure( + request=request, + call_id=str(uuid.uuid4()), + error=error, + operation=operation, + stats=stats, + class_prefixed_error=class_prefixed_error, + ) + + class TelemetryMW: - """洋葱最外层: 观测尝试层看不见的路径,任何路径都留痕(遥测必录)。""" + """洋葱最外层: 只观测尝试层看不见的**缓存命中**。 + + 1.3.5 起不再在此写终态失败行: 终态由 `GatewayClient.chat` 的公开边界经 + `emit_terminal_once` 统一写出。两处同时写会让同一次失败出两条终态行, + 而下游正是按 `event_kind = 'terminal_failure'` 计失败调用次数的。 + """ def __init__( self, emitter: TelemetryEmitter, now: Callable[[], float] = time.monotonic @@ -474,26 +714,7 @@ class TelemetryMW: self._now = now async def __call__(self, request: ChatRequest, call_next: CallNext) -> LLMResponse: - started = self._now() - try: - response = await call_next(request) - except (GatewayUnavailableError, GovernanceBackendError, SourceNotConfiguredError) as exc: - await self._emitter.emit_terminal_failure( - request=request, - call_id=str(uuid.uuid4()), - latency_ms=int((self._now() - started) * 1000), - error=str(exc), - ) - raise - except asyncio.CancelledError: - # 尽力而为: 取消也留痕(§5.1 约定④);随后立即重抛 - await self._emitter.emit_terminal_failure( - request=request, - call_id=str(uuid.uuid4()), - latency_ms=int((self._now() - started) * 1000), - error="cancelled", - ) - raise + response = await call_next(request) if response.cache_hit: - await self._emitter.emit_cache_hit(request=request, response=response) + await self._emitter.emit_cache_hit(request=request, response=response, operation="chat") return response diff --git a/src/polygateway/ocr.py b/src/polygateway/ocr.py index 683649f..501ae18 100644 --- a/src/polygateway/ocr.py +++ b/src/polygateway/ocr.py @@ -37,7 +37,7 @@ from polygateway.middleware.admission import SourceAdmission, settle_and_release from polygateway.middleware.breaker import BreakerGate from polygateway.middleware.ratelimit import QuotaGate from polygateway.middleware.retry import StallClock, _failure_reason, backoff_delay -from polygateway.middleware.telemetry import TelemetryEmitter +from polygateway.middleware.telemetry import TelemetryEmitter, emit_terminal_once from polygateway.ports import OutcomeAwareSelector from polygateway.types import ( CallStats, @@ -67,6 +67,7 @@ if TYPE_CHECKING: ) from polygateway.types import ( BackpressurePolicy, + CallOperation, OcrLayoutTransportResult, OcrTextTransportResult, RetryPolicy, @@ -128,7 +129,9 @@ class OcrClient: self._breaker = BreakerGate(breaker, scope=self._scope) self._transport = transport self._retry = retry - self._emitter = TelemetryEmitter(telemetry, text_cap=text_cap) if telemetry else None + self._emitter = ( + TelemetryEmitter(telemetry, scope=self._scope, text_cap=text_cap) if telemetry else None + ) self._telemetry = telemetry # 限流/熔断后端在此之外只以 QuotaGate/BreakerGate 的形态存在,自持一份 # 引用才关得到自建的 redis 客户端(设计 §3.4) @@ -179,7 +182,13 @@ class OcrClient: tenant_id, meta, origin="recognize_text(tenant_id=..., meta=...)" ) outcome, call_stats = await self._call( - "text", image, session_id, parent_call_id, dimension_tenant_id, dimensions + "text", + "recognize_text", + image, + session_id, + parent_call_id, + dimension_tenant_id, + dimensions, ) result = outcome.result return OcrTextResult( @@ -210,7 +219,13 @@ class OcrClient: tenant_id, meta, origin="parse_layout(tenant_id=..., meta=...)" ) outcome, call_stats = await self._call( - "layout", image, session_id, parent_call_id, dimension_tenant_id, dimensions + "layout", + "parse_layout", + image, + session_id, + parent_call_id, + dimension_tenant_id, + dimensions, ) result = outcome.result return OcrLayoutResult( @@ -241,6 +256,7 @@ class OcrClient: async def _call( self, kind: _OcrKind, + operation: CallOperation, image: bytes, session_id: str | None, parent_call_id: str | None, @@ -254,6 +270,46 @@ class OcrClient: # M1 例外: `image` 校验在 `_call` 内而非公开方法,故上下文在该校验 # **通过之后**创建——这样设计 §3 的"校验在统计边界外"对 OCR 才成立 context = _CallContext(now=self._now) + try: + return await self._run( + kind, operation, image, session_id, parent_call_id, tenant_id, meta, context + ) + except PolyGatewayError as exc: + await self._emit_terminal( + kind, operation, image, session_id, parent_call_id, tenant_id, meta, context, exc + ) + raise + except asyncio.CancelledError: + # 三条链路同一口径尽力写一条(允许 0 条);取消优先,不 shield + await self._emit_terminal( + kind, + operation, + image, + session_id, + parent_call_id, + tenant_id, + meta, + context, + "cancelled", + ) + raise + + async def _run( + self, + kind: _OcrKind, + operation: CallOperation, + image: bytes, + session_id: str | None, + parent_call_id: str | None, + tenant_id: str | None, + meta: dict[str, Any], + context: _CallContext, + ) -> tuple[_AttemptOutcome, CallStats]: + """选源与重试循环。 + + 无源的 raise 必须在本方法内(而非循环之前的调用方): 它得被 `_call` 的 + `try` 包住,否则无源终态行根本写不出来(设计 §3.5)。 + """ if not self._sources: raise AllSourcesExhausted(scope=self._scope, reason="no_sources", retry_after_s=0.0) fails = 0 @@ -268,6 +324,7 @@ class OcrClient: async with clock.attempting(): outcome = await self._attempt( kind, + operation, image, *picked, reasons, @@ -293,6 +350,7 @@ class OcrClient: async def _attempt( self, kind: _OcrKind, + operation: CallOperation, image: bytes, source: SourceConfig, permit: Permit, @@ -318,6 +376,7 @@ class OcrClient: latency_ms = int((self._now() - started) * 1000) await self._emit( kind, + operation, image, source, call_id, @@ -326,6 +385,7 @@ class OcrClient: parent_call_id, tenant_id, meta, + context, result, ) return _AttemptOutcome(result, source, call_id, latency_ms) @@ -333,6 +393,7 @@ class OcrClient: await self._gate_on_terminal(exc, entry) await self._emit( kind, + operation, image, source, call_id, @@ -341,6 +402,7 @@ class OcrClient: parent_call_id, tenant_id, meta, + context, error=exc, ) raise @@ -349,6 +411,7 @@ class OcrClient: await self._record_quietly(self._breaker.release_probe(entry)) await self._emit( kind, + operation, image, source, call_id, @@ -357,6 +420,7 @@ class OcrClient: parent_call_id, tenant_id, meta, + context, error="cancelled", ) raise @@ -368,6 +432,7 @@ class OcrClient: self._feed_outcome(source.name, ok=False) await self._emit( kind, + operation, image, source, call_id, @@ -376,6 +441,7 @@ class OcrClient: parent_call_id, tenant_id, meta, + context, error=exc, ) return _FailedAttempt(exc, immediate=dead) @@ -417,9 +483,60 @@ class OcrClient: except (GovernanceBackendError, SourceNotConfiguredError) as exc: logger.warning("OCR 治理记账写回降级(不冒泡): {}", exc) + def _request_for( + self, + kind: _OcrKind, + image: bytes, + session_id: str | None, + parent_call_id: str | None, + tenant_id: str | None, + meta: dict[str, Any], + context: _CallContext, + ) -> ChatRequest: + """OCR 行的现场 ChatRequest: 占位摘要,**图像 bytes 永不入库**。 + + 尝试行与终态行共用同一个构造点: 占位字面量复制成两份就会漂移。 + 调用方维度与上下文必须显式填回(OCR 不走 chat 洋葱),否则 OCR 行的 + 维度恒为空、`logical_call_id` 恒为 NULL。 + """ + return ChatRequest( + messages=[{"role": "user", "content": f""}], + session_id=session_id, + parent_call_id=parent_call_id, + tenant_id=tenant_id, + meta=meta, + call_context=context, + ) + + async def _emit_terminal( + self, + kind: _OcrKind, + operation: CallOperation, + image: bytes, + session_id: str | None, + parent_call_id: str | None, + tenant_id: str | None, + meta: dict[str, Any], + context: _CallContext, + error: PolyGatewayError | str, + ) -> None: + """终态行: 沿用 `` 占位,错误文本保留类名前缀。""" + await emit_terminal_once( + self._emitter, + request=self._request_for( + kind, image, session_id, parent_call_id, tenant_id, meta, context + ), + context=context, + error=error, + operation=operation, + # metric ocr-call-success 的注册口径是按类名归组,终态行同款保留 + class_prefixed_error=True, + ) + async def _emit( self, kind: _OcrKind, + operation: CallOperation, image: bytes, source: SourceConfig, call_id: str, @@ -428,20 +545,15 @@ class OcrClient: parent_call_id: str | None, tenant_id: str | None, meta: dict[str, Any], + context: _CallContext, result: OcrTextTransportResult | OcrLayoutTransportResult | None = None, - error: object | None = None, + error: PolyGatewayError | str | None = None, ) -> None: """逐尝试遥测(单一 Emitter): messages 占位摘要,图像 bytes 绝不入库。""" if self._emitter is None: return - # 这个 ChatRequest 只为复用同一个 Emitter 而现场构造(OCR 不走 chat 洋葱), - # 故调用方维度必须在这里显式填回,否则 OCR 行的维度恒为空 - request = ChatRequest( - messages=[{"role": "user", "content": f""}], - session_id=session_id, - parent_call_id=parent_call_id, - tenant_id=tenant_id, - meta=meta, + request = self._request_for( + kind, image, session_id, parent_call_id, tenant_id, meta, context ) latency_ms = int((self._now() - started) * 1000) response = None @@ -461,20 +573,19 @@ class OcrClient: source_name=source.name, usage_source="measured", ) - # 错误带异常类名前缀(metric ocr-call-success 注册口径: 按类名归组) - if error is None or isinstance(error, str): - error_text = error - else: - error_text = f"{type(error).__name__}: {error}" + # 错误文本的类名前缀现由出口的显式策略参数承担(设计 §6 I7): + # 三处各拼一遍才是下一次漂移的种子,而取消行传的是字符串,不受前缀影响 await self._emitter.emit_attempt( request=request, source=source, call_id=call_id, latency_ms=latency_ms, response=response, - error=error_text, + error=error, # OCR 走 MonkeyOCR 自有端点,没有推理参数可言(理由同 embedding) reasoning_applies=False, + operation=operation, + class_prefixed_error=True, ) @staticmethod diff --git a/src/polygateway/ports.py b/src/polygateway/ports.py index ed5b20b..a8397d8 100644 --- a/src/polygateway/ports.py +++ b/src/polygateway/ports.py @@ -271,7 +271,7 @@ class TelemetryStatusProvider(Protocol): @runtime_checkable class TelemetryRecorder(Protocol): - """遥测后端;26 字段冻结(M1 设计 §4.4 + issue #3/#4/#11/#16/#20),唯一调用点是 TelemetryEmitter。 + """遥测后端;36 字段冻结(M1 设计 §4.4 + issue #3/#4/#11/#16/#20 + 1.3.5),唯一调用点是 TelemetryEmitter。 新增参数不设默认值: 库外无第三方实现者(三项目迁移时删除了各自的同名 Protocol),完整签名的成本为零,而少写一列会被 emitter 的降级吞成 warning。 @@ -286,6 +286,15 @@ class TelemetryRecorder(Protocol): recorder 只负责落库,不做任何语义判断,与 `sampling` 列由 `canonical_sampling_json()` 在 emitter 侧定型是同一先例。 + + 1.3.5 新增十列同理已在 emitter 侧定型: `operation` 是**公开方法**四值之一 + (与 `PolyGatewayError.operation` 这个 HTTP 子操作是两个语义);`event_kind` 区分 + attempt / cache_hit / terminal_failure 三类行;`attempts` 与 `total_latency_ms` + 只在终态行非空;`error_body` 沿用 transport 侧 `summarize_body` 的上限, + **不进 `PGW_TELEMETRY_TEXT_CAP` 的覆盖面**。 + + **本签名是装配闸的唯一事实源**: `TelemetryEmitter.__init__` 按它派生参数名做 + 一次 `signature.bind` 形状校验(设计 §7),改本签名即改闸的判据。 """ async def record_llm_call( @@ -317,4 +326,14 @@ class TelemetryRecorder(Protocol): meta: str, thinking_observation: str, reasoning_effort: str | None, + scope: str, + operation: str, + logical_call_id: str | None, + event_kind: str, + http_status_code: int | None, + error_type: str | None, + cause_type: str | None, + error_body: str | None, + attempts: int | None, + total_latency_ms: int | None, ) -> None: ... diff --git a/src/polygateway/telemetry/schema.py b/src/polygateway/telemetry/schema.py index 46ada8d..1186e36 100644 --- a/src/polygateway/telemetry/schema.py +++ b/src/polygateway/telemetry/schema.py @@ -5,8 +5,8 @@ 多处各存一份必然漂移,而漂移的表现是"下游照打印的 SQL 建完表,库仍报缺列"。 **`COLUMNS` 是 INSERT 字段序,不是物理列序**: 数据库自填的 `created_at` 不在其中(它带 -`DEFAULT now()` / `datetime('now')`,库从不显式写它)。物理表列 = 26 个 INSERT 字段 + -`created_at` = 27;列数断言一律按物理列数写,两套口径混用是最易错处。 +`DEFAULT now()` / `datetime('now')`,库从不显式写它)。物理表列 = 36 个 INSERT 字段 + +`created_at` = 37;列数断言一律按物理列数写,两套口径混用是最易错处。 本模块只依赖标准库: `telemetry/` 与 `backends/`、`transports/`、`structured/` 同层且 互不依赖(import-linter 契约执法)。 @@ -52,7 +52,17 @@ CREATE TABLE IF NOT EXISTS llm_calls ( tenant_id TEXT NOT NULL DEFAULT '', meta TEXT NOT NULL DEFAULT '{}', thinking_observation TEXT, - reasoning_effort TEXT + reasoning_effort TEXT, + scope TEXT, + operation TEXT, + logical_call_id TEXT, + event_kind TEXT, + http_status_code INTEGER, + error_type TEXT, + cause_type TEXT, + error_body TEXT, + attempts INTEGER, + total_latency_ms INTEGER ); """ @@ -84,7 +94,17 @@ CREATE TABLE IF NOT EXISTS llm_calls ( tenant_id TEXT NOT NULL DEFAULT '', meta JSONB NOT NULL DEFAULT '{}'::jsonb, thinking_observation TEXT, - reasoning_effort TEXT + reasoning_effort TEXT, + scope TEXT, + operation TEXT, + logical_call_id TEXT, + event_kind TEXT, + http_status_code INTEGER, + error_type TEXT, + cause_type TEXT, + error_body TEXT, + attempts INTEGER, + total_latency_ms INTEGER ); """ @@ -105,6 +125,18 @@ SQLITE_BACKFILL = ( # 同样可空,但这里 NULL 表达的是"调用方没表态"(issue #20): 它与 'none' # (明确要求不推理)是两回事,折叠成任一档都等于替上游声称了它没说过的事 ("reasoning_effort", "TEXT"), + # 1.3.5 十列: 全部可空且无默认值——旧行的 NULL 表达的是"补列之前根本没记过 + # 这件事",与任何哨兵值都不是一回事,故不回填(设计 §5) + ("scope", "TEXT"), + ("operation", "TEXT"), + ("logical_call_id", "TEXT"), + ("event_kind", "TEXT"), + ("http_status_code", "INTEGER"), + ("error_type", "TEXT"), + ("cause_type", "TEXT"), + ("error_body", "TEXT"), + ("attempts", "INTEGER"), + ("total_latency_ms", "INTEGER"), ) # PG 补列的列定义。语句由此派生成两份文本(见下),使"库内执行的那份"与"打印给 @@ -120,6 +152,17 @@ _PG_BACKFILL_DECLS = ( # 可空,理由同 SQLITE_BACKFILL 同名项 ("thinking_observation", "TEXT"), ("reasoning_effort", "TEXT"), + # 1.3.5 十列,列序与 SQLITE_BACKFILL 逐项对齐(两条路径的物理列序不许分叉) + ("scope", "TEXT"), + ("operation", "TEXT"), + ("logical_call_id", "TEXT"), + ("event_kind", "TEXT"), + ("http_status_code", "INTEGER"), + ("error_type", "TEXT"), + ("cause_type", "TEXT"), + ("error_body", "TEXT"), + ("attempts", "INTEGER"), + ("total_latency_ms", "INTEGER"), ) # 新列排在 created_at 之后: 与旧表 ALTER 追加的位置一致(见 SQLITE_BACKFILL 同款注释)。 @@ -158,6 +201,17 @@ COLUMNS = ( "meta", "thinking_observation", "reasoning_effort", + # —— 1.3.5 逻辑调用统计与结构化失败诊断(issue #19/#23)—— + "scope", + "operation", + "logical_call_id", + "event_kind", + "http_status_code", + "error_type", + "cause_type", + "error_body", + "attempts", + "total_latency_ms", ) _COLUMN_SET = frozenset(COLUMNS) diff --git a/src/polygateway/transports/openai_compat.py b/src/polygateway/transports/openai_compat.py index a76673a..55f79ea 100644 --- a/src/polygateway/transports/openai_compat.py +++ b/src/polygateway/transports/openai_compat.py @@ -154,18 +154,28 @@ def _classify(status: int) -> tuple[type[PolyGatewayError], str]: def _status_to_error( - source: SourceConfig, status: int, body_text: str, headers: Mapping[str, str] + source: SourceConfig, + status: int, + body_text: str, + headers: Mapping[str, str], + *, + operation: str, ) -> Exception: """非 2xx → 领域错误,**全部分支**携带响应体摘要(issue #10)。 摘要只算一次,message 与 `body_text` 共用同一份串: 两份不同长度会让"遥测里 看到的"与"下游 catch 到的"对不上,排查时反而多一层困惑。 + + `operation` 是 **HTTP 子操作**词表(`chat` / `embedding` / `ocr_text` / ...),由 + 调用点显式给定。它曾被硬编码成 `"chat"`,而 `embed()` 的非 200 分支也走它 + ——于是现存所有 embedding HTTP 失败的 `exc.operation` 都是错的(1.3.5 设计 §5 I1)。 + 注意它与遥测新列 `operation`(公开方法四值)是**两个语义**,不做自动转换。 """ summary = summarize_body(body_text) ctx: dict[str, Any] = { "source_name": source.name, "status_code": status, - "operation": "chat", + "operation": operation, "body_text": summary, } if status == 429: @@ -509,7 +519,10 @@ class OpenAICompatTransport: except httpx.TransportError as exc: raise TransientError(f"{source.name} 网络错误: {exc}", **ctx) from exc if resp.status_code != 200: - raise _status_to_error(source, resp.status_code, resp.text, resp.headers) + # 历史误标修正: 本分支属 `embed()`,与上方 ctx 同为 `"embedding"` + raise _status_to_error( + source, resp.status_code, resp.text, resp.headers, operation="embedding" + ) return _parse_embedding_payload(resp, source, len(texts)) async def _complete_stream( @@ -524,7 +537,9 @@ class OpenAICompatTransport: async with client.stream("POST", url, json=payload) as resp: if resp.status_code != 200: body = (await resp.aread()).decode("utf-8", errors="replace") - raise _status_to_error(source, resp.status_code, body, resp.headers) + raise _status_to_error( + source, resp.status_code, body, resp.headers, operation="chat" + ) sink: dict[str, Any] = {} guarded = stream_with_liveness_timeouts( _iter_sse_deltas(resp.aiter_lines(), sink), @@ -622,7 +637,9 @@ class OpenAICompatTransport: """非流式快路径(三项目均无,库新增): 单 JSON 响应,仅 total 超时。""" resp = await client.post(url, json=payload) if resp.status_code != 200: - raise _status_to_error(source, resp.status_code, resp.text, resp.headers) + raise _status_to_error( + source, resp.status_code, resp.text, resp.headers, operation="chat" + ) try: body = resp.json() except json.JSONDecodeError as exc: diff --git a/tests/unit/test_cache.py b/tests/unit/test_cache.py index 6e86044..113e6dd 100644 --- a/tests/unit/test_cache.py +++ b/tests/unit/test_cache.py @@ -570,7 +570,7 @@ class TestTelemetryCapDoesNotPoisonTheCacheKey: before = build_cache_key("m", messages, "proj", None) rec = self._Rows() - await TelemetryEmitter(rec, text_cap=8).emit_attempt( + await TelemetryEmitter(rec, text_cap=8, scope="LLM").emit_attempt( request=ChatRequest(messages=messages), source=SourceConfig( name="s1", @@ -585,6 +585,7 @@ class TestTelemetryCapDoesNotPoisonTheCacheKey: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) # 截断确实发生了(否则本用例恒真) logged = json.loads(rec.rows[0]["messages"]) diff --git a/tests/unit/test_client.py b/tests/unit/test_client.py index bc77f2c..4213097 100644 --- a/tests/unit/test_client.py +++ b/tests/unit/test_client.py @@ -13,6 +13,7 @@ from polygateway import ( GatewayClient, GatewaySettings, RequestRejectedError, + ResultInvalidError, gather_bounded, ) from polygateway.backends.memory.breaker import InMemoryGate @@ -25,6 +26,7 @@ from polygateway.transports.openai_compat import OpenAICompatTransport from polygateway.types import ( BackpressurePolicy, BreakerConfig, + ChatRequest, Effort, GlobalLimits, RetryPolicy, @@ -846,11 +848,19 @@ _CACHE_ENV = dict( class _Closable: - """记 close 次数的假组件;所有权纪律的唯一观测点。""" + """记 close 次数的假组件;所有权纪律的唯一观测点。 + + 带 `record_llm_call(**fields)` 是因为它也被当作注入的 telemetry recorder 用: + 1.3.5 的装配闸在构造期就会拒掉不满足 `TelemetryRecorder` 的对象 + (否则下游升级后 100% 丢遥测而调用照常成功)。`**fields` 形态天然兼容。 + """ def __init__(self): self.closed = 0 + async def record_llm_call(self, **fields): + pass + async def aclose(self): self.closed += 1 @@ -861,6 +871,9 @@ class _SyncClosable: def __init__(self): self.closed = 0 + async def record_llm_call(self, **fields): + pass + def close(self): self.closed += 1 @@ -1499,3 +1512,136 @@ class TestLogicalCallStats: async with _client() as client: with pytest.raises(ValueError, match="meta"): await client.chat(self._MSG, meta={"BAD-KEY": 1}) + + +class TestChatTerminalFailureRows: + """chat 链路的**真实**终态行(1.3.5 设计 §6;补漏而非改口径)。 + + 这些路径改前一条失败行都没有(结构化耗尽)或只有尝试行, + "这次调用到底失败了几次"因此 SQL 答不出来。 + """ + + _MSG = [{"role": "user", "content": "hi"}] + + def _rows(self, recorder, kind): + return [r for r in recorder.rows if r["event_kind"] == kind] + + async def test_structured_exhaustion_writes_the_only_failure_row(self): + """结构化耗尽发生在 transport 成功之后: attempt 行全是成功行,终态是唯一记录。""" + from pydantic import BaseModel + + class Answer(BaseModel): + answer: int + + recorder = _MemoryRecorder() + async with _client( + handler=lambda request: _sse("not json at all"), + telemetry=recorder, + structured_max_retries=1, + ) as client: + with pytest.raises(ResultInvalidError): + await client.chat(self._MSG, structured=Answer) + attempts = self._rows(recorder, "attempt") + terminals = self._rows(recorder, "terminal_failure") + assert [a["error"] for a in attempts] == [None, None] # 两次尝试都成功 + assert len(terminals) == 1 + row = terminals[0] + assert row["error_type"] == "ResultInvalidError" + # C2: 有界结构化说明并入 error,且不含 raw_text(正文预算已由 attempt 行承担) + assert "validation=" in row["error"] or "repair=" in row["error"] + assert len(row["error"]) < 1200 + + async def test_retry_exhaustion_writes_exactly_one_terminal_row(self): + """重试耗尽: 逐次 attempt 行之外只能有**一条**终态行。 + + 终态行的 `attempts` 是整次逻辑调用的真实尝试数——这正是改前 SQL + 答不出的"这次调用到底重试了几次"。 + """ + recorder = _MemoryRecorder() + async with _client( + handler=lambda request: httpx.Response(503), + telemetry=recorder, + retry=RetryPolicy(3, 0.001, 0.01), + ) as client: + with pytest.raises(AllSourcesExhausted): + await client.chat(self._MSG) + attempts = self._rows(recorder, "attempt") + terminals = self._rows(recorder, "terminal_failure") + assert len(attempts) == 3 + assert len(terminals) == 1 + row = terminals[0] + assert row["error_type"] == "AllSourcesExhausted" + assert row["attempts"] == 3 # 整次逻辑调用的尝试数 + assert row["scope"] == "llm" and row["operation"] == "chat" + # 终态行的 latency_ms 与 total_latency_ms 同取一份冻结快照 + assert row["latency_ms"] == row["total_latency_ms"] + + async def test_request_rejected_now_has_both_an_attempt_and_a_terminal_row(self): + """已批准的下游可见变化: 400 密集负载的错误行翻倍,失败计数只能取终态。""" + recorder = _MemoryRecorder() + async with _client( + handler=lambda request: httpx.Response(400, json={"error": {"message": "bad"}}), + telemetry=recorder, + ) as client: + with pytest.raises(RequestRejectedError): + await client.chat(self._MSG) + attempts = self._rows(recorder, "attempt") + terminals = self._rows(recorder, "terminal_failure") + assert len(attempts) == 1 and attempts[0]["http_status_code"] == 400 + assert len(terminals) == 1 + # C1: 终态不搬运最后一次 attempt 的状态码与正文 + assert terminals[0]["http_status_code"] is None + assert terminals[0]["error_body"] is None + # 两类行共享同一 logical_call_id,归因查询才连得起来 + assert attempts[0]["logical_call_id"] == terminals[0]["logical_call_id"] + + async def test_non_domain_exception_writes_no_terminal_row(self): + """编程错原样传播,本版**不承诺**任何统计或终态行,也不偷偷改分类。""" + recorder = _MemoryRecorder() + + async def boom(request): + raise KeyError("programming error") + + async with _client(telemetry=recorder) as client: + client._handler = boom + with pytest.raises(KeyError): + await client.chat(self._MSG) + assert self._rows(recorder, "terminal_failure") == [] + + async def test_cancellation_writes_at_most_one_terminal_row(self): + """取消尽力写一条(允许 0),且 `CancelledError` 类型与语义不变。""" + recorder = _MemoryRecorder() + + async def hang(request): + raise asyncio.CancelledError + + async with _client(telemetry=recorder) as client: + client._handler = hang + with pytest.raises(asyncio.CancelledError): + await client.chat(self._MSG) + terminals = self._rows(recorder, "terminal_failure") + assert len(terminals) == 1 + assert terminals[0]["error"] == "cancelled" + # 字符串不解析猜诊断 + assert terminals[0]["error_type"] is None + + async def test_terminal_row_is_written_once_per_logical_call(self): + """`claim_terminal` 去重: 同一次调用即便出口被多次触达也只有一条。""" + from polygateway.middleware.telemetry import emit_terminal_once + from polygateway.types import _CallContext + + recorder = _MemoryRecorder() + async with _client(telemetry=recorder) as client: + context = _CallContext(now=client._now) + request = ChatRequest(messages=self._MSG, call_context=context) + for _ in range(3): + await emit_terminal_once( + client._emitter, + request=request, + context=context, + error=AllSourcesExhausted( + scope="llm", reason="retry_exhausted", retry_after_s=1.0 + ), + operation="chat", + ) + assert len(self._rows(recorder, "terminal_failure")) == 1 diff --git a/tests/unit/test_embedding.py b/tests/unit/test_embedding.py index 3384a75..f6ad5a3 100644 --- a/tests/unit/test_embedding.py +++ b/tests/unit/test_embedding.py @@ -580,7 +580,12 @@ class TestReasonlessTelemetryContract: client, _ = _embed_client([_src(enable_thinking=True)], script, telemetry=recorder) with pytest.raises(AllSourcesExhausted if exhausted else RequestRejectedError): await client.embed(["text"]) - assert len(recorder.rows) == len(script) + # 1.3.5: 逐次 attempt 行之外,本次逻辑调用另有**一条**终态行 + attempts = [r for r in recorder.rows if r["event_kind"] == "attempt"] + terminals = [r for r in recorder.rows if r["event_kind"] == "terminal_failure"] + assert len(attempts) == len(script) + assert len(terminals) == 1 + # 推理档在三类行上都必须是 NULL: embed payload 从不带推理参数 assert all(r["error"] and r["reasoning_effort"] is None for r in recorder.rows) async def test_embedding_wire_ignores_reasoning_configuration(self): diff --git a/tests/unit/test_ocr_client.py b/tests/unit/test_ocr_client.py index acac9c2..d4f03d0 100644 --- a/tests/unit/test_ocr_client.py +++ b/tests/unit/test_ocr_client.py @@ -613,7 +613,13 @@ class TestReasonlessTelemetryContract: client, _, _ = _client([_src(enable_thinking=True)], script, telemetry=recorder) with pytest.raises(AllSourcesExhausted if exhausted else RequestRejectedError): await getattr(client, method)(b"image") - assert len(recorder.rows) == len(script) + # 1.3.5: 逐次 attempt 行之外,本次逻辑调用另有**一条**终态行 + attempts = [r for r in recorder.rows if r["event_kind"] == "attempt"] + terminals = [r for r in recorder.rows if r["event_kind"] == "terminal_failure"] + assert len(attempts) == len(script) + assert len(terminals) == 1 + # 终态行的 operation 是**公开方法**名,与尝试行一致 + assert terminals[0]["operation"] == method assert all(r["error"] and r["reasoning_effort"] is None for r in recorder.rows) diff --git a/tests/unit/test_openai_compat.py b/tests/unit/test_openai_compat.py index f54f796..b65f0c4 100644 --- a/tests/unit/test_openai_compat.py +++ b/tests/unit/test_openai_compat.py @@ -116,7 +116,7 @@ async def _recorded_cost(result, source): source_name=source.name, usage_source=result.usage_source, ) - await TelemetryEmitter(recorder, pricing=_PRICING, text_cap=None).emit_attempt( + await TelemetryEmitter(recorder, pricing=_PRICING, text_cap=None, scope="LLM").emit_attempt( request=ChatRequest(messages=[{"role": "user", "content": "hi"}]), source=source, call_id="cid-1", @@ -124,6 +124,7 @@ async def _recorded_cost(result, source): response=response, error=None, reasoning_applies=True, + operation="chat", ) return recorder.rows[0]["cost"] diff --git a/tests/unit/test_ports.py b/tests/unit/test_ports.py index f772271..57beecc 100644 --- a/tests/unit/test_ports.py +++ b/tests/unit/test_ports.py @@ -124,6 +124,18 @@ class _DummyRecorder: reasoning_tokens, tenant_id, meta, + thinking_observation, + reasoning_effort, + scope, + operation, + logical_call_id, + event_kind, + http_status_code, + error_type, + cause_type, + error_body, + attempts, + total_latency_ms, ) -> None: ... @@ -275,8 +287,47 @@ class TestTelemetryRecorderSignature: params = inspect.signature(TelemetryRecorder.record_llm_call).parameters assert {"tenant_id", "meta"} <= set(params) + def test_call_observability_fields_are_declared(self): + """1.3.5 十列进协议(issue #19/#23);字段总数以实测为准不凭记忆。 + + 本签名同时是装配闸的事实源(`_assert_recorder_shape` 按它派生参数名), + 故它与实现一旦漂移,下游自定义 recorder 会在装配期就被拒。 + """ + import inspect + + params = inspect.signature(TelemetryRecorder.record_llm_call).parameters + assert { + "scope", + "operation", + "logical_call_id", + "event_kind", + "http_status_code", + "error_type", + "cause_type", + "error_body", + "attempts", + "total_latency_ms", + } <= set(params) + assert len(params) - 1 == 36 # 减掉 self + @pytest.mark.parametrize( - "name", ["tenant_id", "meta", "thinking_observation", "reasoning_effort"] + "name", + [ + "tenant_id", + "meta", + "thinking_observation", + "reasoning_effort", + "scope", + "operation", + "logical_call_id", + "event_kind", + "http_status_code", + "error_type", + "cause_type", + "error_body", + "attempts", + "total_latency_ms", + ], ) def test_caller_dimensions_have_no_default(self, name): import inspect diff --git a/tests/unit/test_pricing.py b/tests/unit/test_pricing.py index de031a1..ee6690d 100644 --- a/tests/unit/test_pricing.py +++ b/tests/unit/test_pricing.py @@ -169,7 +169,7 @@ def _source(model="qwen-max"): class TestEmitterCost: async def test_success_row_costed(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None) + emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(), @@ -178,18 +178,19 @@ class TestEmitterCost: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] == pytest.approx(7.2) async def test_cache_hit_row_costs_zero(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None) - await emitter.emit_cache_hit(request=_REQ, response=_resp(cache_hit=True)) + emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None, scope="LLM") + await emitter.emit_cache_hit(request=_REQ, response=_resp(cache_hit=True), operation="chat") assert rec.rows[0]["cost"] == 0.0 async def test_failure_row_cost_none(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None) + emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(), @@ -198,12 +199,13 @@ class TestEmitterCost: response=None, error="TransientError: boom", reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] is None async def test_unknown_model_none_without_blocking(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None) + emitter = TelemetryEmitter(rec, pricing=_TABLE, text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(model="mystery"), @@ -212,13 +214,14 @@ class TestEmitterCost: response=_resp(model="mystery"), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] is None async def test_no_pricing_keeps_none(self): """未注入价格表 = M1 现状: cost 恒 None(回归)。""" rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(), @@ -227,5 +230,6 @@ class TestEmitterCost: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] is None diff --git a/tests/unit/test_telemetry.py b/tests/unit/test_telemetry.py index 49aacca..75999d7 100644 --- a/tests/unit/test_telemetry.py +++ b/tests/unit/test_telemetry.py @@ -8,13 +8,20 @@ import sqlite3 import subprocess from pathlib import Path +import httpx import pytest from loguru import logger from polygateway.backends.memory.breaker import InMemoryGate from polygateway.backends.memory.limiter import InMemoryLimiter from polygateway.embedding import EmbeddingClient -from polygateway.errors import CircuitOpenError, RequestRejectedError +from polygateway.errors import ( + AllSourcesExhausted, + CircuitOpenError, + RequestRejectedError, + ResultInvalidError, + TransientError, +) from polygateway.middleware.cache import digest_messages from polygateway.middleware.telemetry import TelemetryEmitter, TelemetryMW from polygateway.ocr import OcrClient @@ -24,6 +31,7 @@ from polygateway.telemetry.sqlite import SQLiteRecorder from polygateway.types import ( BackpressurePolicy, BreakerConfig, + CallStats, ChatRequest, Effort, EmbeddingTransportResult, @@ -37,6 +45,11 @@ from polygateway.types import ( _REQ = ChatRequest(messages=[{"role": "user", "content": "hi"}], session_id="sess-1") +# 终态行的快照入参(1.3.5): `emit_terminal_failure` 不再收 `latency_ms`, +# `latency_ms` 与 `total_latency_ms` 同取这一份冻结快照,避免双时钟微差。 +# 取 1ms 是为与迁移前逐字写死的 `latency_ms=1` 保持同值。 +_MIGRATED_STATS = CallStats(logical_call_id="lcid-mig", attempts=1, total_latency_ms=1) + _EXPECTED_COLUMNS = [ "call_id", "parent_call_id", @@ -65,8 +78,26 @@ _EXPECTED_COLUMNS = [ "meta", "thinking_observation", "reasoning_effort", + # —— 1.3.5 逻辑调用统计与结构化失败诊断的十列 —— + "scope", + "operation", + "logical_call_id", + "event_kind", + "http_status_code", + "error_type", + "cause_type", + "error_body", + "attempts", + "total_latency_ms", ] +# 1.3.5 之前的 26 个 INSERT 字段(旧表形态),供冻结 INSERT 语句与旧表补列用例复用 +_PRE_135_COLUMNS = [c for c in _EXPECTED_COLUMNS if c != "created_at"][:26] + +# `_PRE_TENANT_DDL` 那张旧表的 23 个物理列(22 个 INSERT 字段 + created_at)。 +# 写成固定切片而非 `[:-N]`: 后者会随每一次补列静默漂移到另一张表上去。 +_PRE_TENANT_PHYSICAL = _EXPECTED_COLUMNS[:23] + def _resp(**overrides): base = { @@ -136,6 +167,20 @@ async def _record_minimal(recorder, call_id="c1", **overrides): "thinking_observation": "unknown", # 同理: `Effort` 归一成裸 str,不表态则是 None(与 'low' 必须分得开) "reasoning_effort": None, + # —— 1.3.5 十列: 默认形态即"一次普通尝试行"—— + "scope": "LLM", + "operation": "chat", + # 库内现场构造的请求没有上下文 → NULL,不造 ID + "logical_call_id": None, + "event_kind": "attempt", + # 诊断四列只在失败的 attempt 行上非空;成功行不统一填 200 + "http_status_code": None, + "error_type": None, + "cause_type": None, + "error_body": None, + # 逻辑快照两列只属终态行 + "attempts": None, + "total_latency_ms": None, } fields.update(overrides) await recorder.record_llm_call(**fields) @@ -217,7 +262,7 @@ class TestSchemaModule: # COLUMNS 是 INSERT 字段序,不含数据库自填的 created_at assert list(COLUMNS) == [c for c in _EXPECTED_COLUMNS if c != "created_at"] - assert len(COLUMNS) == 26 + assert len(COLUMNS) == 36 # 两端 DDL 的列出现顺序 == 物理列序(created_at 在第 19 位) for ddl in (SQLITE_DDL, PG_DDL): assert _first_occurrence_order(ddl, _EXPECTED_COLUMNS) == _EXPECTED_COLUMNS @@ -231,17 +276,25 @@ class TestSchemaModule: "ALTER TABLE llm_calls ADD COLUMN cached_prompt_tokens INTEGER", ) assert PG_BACKFILL[-1] == ( - "reasoning_effort", - "ALTER TABLE llm_calls ADD COLUMN reasoning_effort TEXT", + "total_latency_ms", + "ALTER TABLE llm_calls ADD COLUMN total_latency_ms INTEGER", ) assert all("IF NOT EXISTS" not in stmt for _, stmt in PG_BACKFILL) def test_insert_sql_reproduces_the_frozen_statements(self): - """`insert_sql(backend, COLUMNS)` 与搬迁前的 `_INSERT` 一致(PG 侧去掉冲突目标)。""" + """搬迁前(1.2.1)的 26 字段 `_INSERT` 逐字可重现(PG 侧去掉冲突目标)。 + + 冻结串是"纯搬迁不改行为"的机械证据,故仍按**当时那 26 列**构造; + 1.3.5 补列后的全量语句另由下一条用例按占位符个数断言。 + """ from polygateway.telemetry.schema import COLUMNS, insert_sql - assert insert_sql("sqlite", COLUMNS) == _FROZEN_SQLITE_INSERT - assert insert_sql("postgres", COLUMNS) == _FROZEN_PG_INSERT + assert insert_sql("sqlite", _PRE_135_COLUMNS) == _FROZEN_SQLITE_INSERT + assert insert_sql("postgres", _PRE_135_COLUMNS) == _FROZEN_PG_INSERT + # 全量 36 列: 占位符随列数增长,且不留空洞 + assert insert_sql("postgres", COLUMNS).count("$") == 36 + assert "$36)" in insert_sql("postgres", COLUMNS) + assert insert_sql("sqlite", COLUMNS).count("?") == 36 # 裁剪列表按位置占位符重新编号,不留空洞 assert insert_sql("postgres", ["call_id", "model"]) == ( "INSERT INTO llm_calls (call_id, model) VALUES ($1, $2) ON CONFLICT DO NOTHING" @@ -318,7 +371,7 @@ class TestBackendColumnParity: """新列只能追加在末尾: 旧表经 ALTER 补列必落末尾,插在中间会让两条路径分叉。""" from polygateway.telemetry.schema import COLUMNS - assert COLUMNS[-4:] == ("tenant_id", "meta", "thinking_observation", "reasoning_effort") + assert COLUMNS[-4:] == ("cause_type", "error_body", "attempts", "total_latency_ms") class TestSQLiteRecorder: @@ -594,7 +647,7 @@ class TestSQLiteCallerDimensionsAcceptance: conn = sqlite3.connect(db) cols = [r[1] for r in conn.execute("PRAGMA table_info(llm_calls)")] - assert cols == _EXPECTED_COLUMNS # 22 → 26 个 recorder 字段(+ created_at 共 27 物理列) + assert cols == _EXPECTED_COLUMNS # 22 → 36 个 recorder 字段(+ created_at 共 37 物理列) rows = dict(conn.execute("SELECT call_id, tenant_id FROM llm_calls").fetchall()) assert rows["new-row"] == "tenant-a" assert rows["old-row"] == "" # 不是 None: NULL 会被 RLS 静默吞掉 @@ -636,7 +689,7 @@ class TestSQLiteCallerDimensionsAcceptance: stale = sqlite3.connect(db) assert [r[1] for r in stale.execute("PRAGMA table_info(llm_calls)")] == ( - _EXPECTED_COLUMNS[:-4] + _PRE_TENANT_PHYSICAL ) # 补列确实没成功,用例不是在只读库上空转 @@ -644,7 +697,7 @@ class TestSQLiteSchemaMode: """issue #13: `auto_migrate` 两档——auto 保持自动补列,manual 只裁剪写入不发 DDL。 列数断言一律按**物理列数**写: 旧表 22 个 INSERT 字段 + `created_at` = 23, - 补齐后 26 + `created_at` = 27。混用 INSERT 字段数与物理列数是本处最易错的地方。 + 补齐后 36 + `created_at` = 37。混用 INSERT 字段数与物理列数是本处最易错的地方。 """ def _physical_columns(self, db: Path) -> list[str]: @@ -683,7 +736,7 @@ class TestSQLiteSchemaMode: assert "ALTER TABLE" in message # 给出可直接执行的补列 SQL async def test_auto_mode_still_upgrades_the_legacy_table(self, tmp_path): - """auto + 同款旧表: 现状回归,补列后物理列数 23 → 27。""" + """auto + 同款旧表: 现状回归,补列后物理列数 23 → 37。""" db = tmp_path / "auto_legacy.db" _make_pre_tenant_db(db) @@ -692,10 +745,10 @@ class TestSQLiteSchemaMode: recorder.close() assert self._physical_columns(db) == _EXPECTED_COLUMNS - assert len(self._physical_columns(db)) == 27 + assert len(self._physical_columns(db)) == 37 async def test_manual_mode_still_creates_a_fresh_table(self, tmp_path): - """manual 只管 ALTER,不管 CREATE: 全新库照建,27 个物理列齐全(设计 §4.2)。""" + """manual 只管 ALTER,不管 CREATE: 全新库照建,37 个物理列齐全(设计 §4.2)。""" db = tmp_path / "manual_fresh.db" recorder = SQLiteRecorder(db, auto_migrate=False) await _record_minimal(recorder, call_id="c-fresh", tenant_id="tenant-a") @@ -921,6 +974,17 @@ class TestPostgresBackfillDiscipline: "meta", "thinking_observation", "reasoning_effort", + # 1.3.5 十列: 稳态的定义随补列一起前移,否则本用例会把"每进程首写抢锁"放行 + "scope", + "operation", + "logical_call_id", + "event_kind", + "http_status_code", + "error_type", + "cause_type", + "error_body", + "attempts", + "total_latency_ms", ] def _recorder(self, conn): @@ -1140,7 +1204,7 @@ class TestEmitterRecorderContract: from polygateway.telemetry.schema import COLUMNS rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="cid-1", @@ -1148,6 +1212,7 @@ class TestEmitterRecorderContract: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) assert set(rec.rows[0]) == set(COLUMNS) @@ -1156,7 +1221,7 @@ class TestEmitterRecorderContract: from polygateway.telemetry.schema import COLUMNS rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") if emit == "attempt": await emitter.emit_attempt( request=_REQ, @@ -1166,12 +1231,17 @@ class TestEmitterRecorderContract: response=None, error="boom", reasoning_applies=True, + operation="chat", ) elif emit == "cache_hit": - await emitter.emit_cache_hit(request=_REQ, response=_resp()) + await emitter.emit_cache_hit(request=_REQ, response=_resp(), operation="chat") else: await emitter.emit_terminal_failure( - request=_REQ, call_id="c", latency_ms=1, error="dead" + request=_REQ, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert set(rec.rows[0]) == set(COLUMNS) @@ -1187,7 +1257,7 @@ class TestEmitterThinkingObservation: async def test_attempt_carries_the_verdict_as_a_plain_string(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1195,6 +1265,7 @@ class TestEmitterThinkingObservation: response=_resp(thinking_observation=ThinkingObservation.OBSERVED), error=None, reasoning_applies=True, + operation="chat", ) value = rec.rows[0]["thinking_observation"] assert value == "observed" @@ -1210,7 +1281,7 @@ class TestEmitterThinkingObservation: 是整行,正是 1.3.0 那次"19 次调用一行未落"的同款形态。 """ rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1218,6 +1289,7 @@ class TestEmitterThinkingObservation: response=_resp(thinking_observation="observed"), error=None, reasoning_applies=True, + operation="chat", ) assert len(rec.rows) == 1, "整行被吞了" value = rec.rows[0]["thinking_observation"] @@ -1236,7 +1308,7 @@ class TestEmitterThinkingObservation: messages: list[str] = [] sink_id = logger.add(messages.append, level="WARNING") try: - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1244,6 +1316,7 @@ class TestEmitterThinkingObservation: response=_resp(thinking_observation="OBSERVED"), # 大小写不符即域外 error=None, reasoning_applies=True, + operation="chat", ) finally: logger.remove(sink_id) @@ -1256,17 +1329,22 @@ class TestEmitterThinkingObservation: async def test_cache_hit_replays_the_recorded_verdict(self): """缓存命中回放历史那次的裁定: 与 model/prompt_tokens 同一口径。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_cache_hit( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_cache_hit( request=_REQ, response=_resp(cache_hit=True, thinking_observation=ThinkingObservation.ABSENT), + operation="chat", ) assert rec.rows[0]["thinking_observation"] == "absent" async def test_terminal_failure_records_unknown(self): """终态失败无响应可言,记 `unknown`——它恰好就是"观测不到",不撒谎。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_terminal_failure( - request=_REQ, call_id="c", latency_ms=1, error="dead" + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_terminal_failure( + request=_REQ, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) value = rec.rows[0]["thinking_observation"] assert value == "unknown" @@ -1275,7 +1353,7 @@ class TestEmitterThinkingObservation: async def test_failed_attempt_records_unknown(self): """失败尝试(response=None)同理: 默认视图即 UNKNOWN。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1283,6 +1361,7 @@ class TestEmitterThinkingObservation: response=None, error="boom", reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["thinking_observation"] == "unknown" @@ -1303,7 +1382,7 @@ class TestEmitterReasoningEffort: 映射的源上恒等——本地跑不开映射的源永远看不出这个错。 """ rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=ChatRequest( messages=[{"role": "user", "content": "hi"}], reasoning_effort=Effort.MEDIUM ), @@ -1313,6 +1392,7 @@ class TestEmitterReasoningEffort: response=_resp(applied_effort=Effort.LOW), error=None, reasoning_applies=True, + operation="chat", ) value = rec.rows[0]["reasoning_effort"] assert value == "low" # 不是 medium: 那一档从未发出去过 @@ -1326,7 +1406,7 @@ class TestEmitterReasoningEffort: "哪一档配错了" 是有用信号,不该被过滤掉。 """ rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(reasoning_effort=Effort.HIGH), call_id="c", @@ -1334,13 +1414,14 @@ class TestEmitterReasoningEffort: response=None, error="boom", reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["reasoning_effort"] == "high" async def test_failed_attempt_resolves_the_syntactic_sugar_too(self): """回落走 `effective_effort` 而非裸读字段: `enable_thinking` 也是表态。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(enable_thinking=True), call_id="c", @@ -1348,30 +1429,33 @@ class TestEmitterReasoningEffort: response=None, error="boom", reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["reasoning_effort"] == "auto" async def test_cache_hit_records_the_request_tier_not_the_replayed_one(self): """命中行没有选中源,故记请求档;与 model/prompt_tokens 的回放口径相反。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_cache_hit( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_cache_hit( request=ChatRequest( messages=[{"role": "user", "content": "hi"}], reasoning_effort=Effort.MEDIUM ), response=_resp(cache_hit=True, applied_effort=Effort.LOW), + operation="chat", ) assert rec.rows[0]["reasoning_effort"] == "medium" async def test_terminal_failure_records_the_request_tier(self): """终态失败可能根本没选出源,源级档位无从谈起。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_terminal_failure( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_terminal_failure( request=ChatRequest( messages=[{"role": "user", "content": "hi"}], reasoning_effort=Effort.XHIGH ), call_id="c", - latency_ms=1, error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) value = rec.rows[0]["reasoning_effort"] assert value == "xhigh" @@ -1384,7 +1468,7 @@ class TestEmitterReasoningEffort: 库并不观测模型内部的默认档,记一个推定值等于把"没看见"说成"发生了"。 """ rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") if emit == "attempt": await emitter.emit_attempt( request=_REQ, @@ -1394,12 +1478,19 @@ class TestEmitterReasoningEffort: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) elif emit == "cache_hit": - await emitter.emit_cache_hit(request=_REQ, response=_resp(cache_hit=True)) + await emitter.emit_cache_hit( + request=_REQ, response=_resp(cache_hit=True), operation="chat" + ) else: await emitter.emit_terminal_failure( - request=_REQ, call_id="c", latency_ms=1, error="dead" + request=_REQ, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert rec.rows[0]["reasoning_effort"] is None @@ -1410,7 +1501,7 @@ class TestEmitterReasoningEffort: 挂上 `auto` ——那一档从来没有、也不可能被发出去。 """ rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(enable_thinking=True), call_id="c", @@ -1418,6 +1509,7 @@ class TestEmitterReasoningEffort: response=None, error="boom", reasoning_applies=False, + operation="chat", ) assert rec.rows[0]["reasoning_effort"] is None @@ -1446,7 +1538,7 @@ class TestEmitterReasoningEffort: messages: list[str] = [] sink_id = logger.add(messages.append, level="WARNING") try: - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1454,6 +1546,7 @@ class TestEmitterReasoningEffort: response=_resp(applied_effort="lowest"), error=None, reasoning_applies=True, + operation="chat", ) finally: logger.remove(sink_id) @@ -1466,7 +1559,7 @@ class TestEmitterReasoningEffort: async def test_a_bare_string_tier_still_lands(self): """裸串在域内时照常归一并落库,整行不得丢失。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1474,6 +1567,7 @@ class TestEmitterReasoningEffort: response=_resp(applied_effort="max"), error=None, reasoning_applies=True, + operation="chat", ) assert len(rec.rows) == 1, "整行被吞了" value = rec.rows[0]["reasoning_effort"] @@ -1486,7 +1580,7 @@ class TestEmitterObservabilityFields: async def test_attempt_carries_the_response_values(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="cid-1", @@ -1494,6 +1588,7 @@ class TestEmitterObservabilityFields: response=_resp(cached_prompt_tokens=64, model_reported="m-real", reasoning_tokens=7), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cached_prompt_tokens"] == 64 assert rec.rows[0]["model_reported"] == "m-real" @@ -1501,7 +1596,7 @@ class TestEmitterObservabilityFields: async def test_failed_attempt_has_no_provider_facts(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="cid-2", @@ -1509,6 +1604,7 @@ class TestEmitterObservabilityFields: response=None, error="boom", reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cached_prompt_tokens"] is None assert rec.rows[0]["model_reported"] is None @@ -1517,9 +1613,10 @@ class TestEmitterObservabilityFields: async def test_cache_hit_replays_the_recorded_values(self): """决策 B1: 命中行原样回放,故命中率统计必须带 WHERE cache_hit = false。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_cache_hit( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_cache_hit( request=_REQ, response=_resp(cached_prompt_tokens=64, model_reported="m-real", reasoning_tokens=7), + operation="chat", ) row = rec.rows[0] assert row["cache_hit"] is True @@ -1528,8 +1625,12 @@ class TestEmitterObservabilityFields: async def test_terminal_failure_records_none(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_terminal_failure( - request=_REQ, call_id="c", latency_ms=1, error="dead" + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_terminal_failure( + request=_REQ, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert rec.rows[0]["cached_prompt_tokens"] is None assert rec.rows[0]["model_reported"] is None @@ -1552,7 +1653,7 @@ class TestEmitterSamplingColumn: async def test_attempt_merges_source_extra_body(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=self._SAMPLED, source=_source(extra_body={"temperature": 0}), call_id="c", @@ -1560,13 +1661,14 @@ class TestEmitterSamplingColumn: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) assert json.loads(rec.rows[0]["sampling"]) == {"seed": 42, "temperature": 0} async def test_response_format_never_leaks_into_the_column(self): """三行都不得出现 response_format——它不是采样参数。""" rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") await emitter.emit_attempt( request=self._SAMPLED, source=_source(), @@ -1575,10 +1677,15 @@ class TestEmitterSamplingColumn: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) - await emitter.emit_cache_hit(request=self._SAMPLED, response=_resp()) + await emitter.emit_cache_hit(request=self._SAMPLED, response=_resp(), operation="chat") await emitter.emit_terminal_failure( - request=self._SAMPLED, call_id="c", latency_ms=1, error="dead" + request=self._SAMPLED, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert len(rec.rows) == 3 for row in rec.rows: @@ -1588,19 +1695,23 @@ class TestEmitterSamplingColumn: async def test_sourceless_entries_record_call_level_only(self, emit): """两个最外层入口没有"生效源"可言,与 model/source_name 置空同一先例。""" rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") if emit == "cache_hit": - await emitter.emit_cache_hit(request=self._SAMPLED, response=_resp()) + await emitter.emit_cache_hit(request=self._SAMPLED, response=_resp(), operation="chat") else: await emitter.emit_terminal_failure( - request=self._SAMPLED, call_id="c", latency_ms=1, error="dead" + request=self._SAMPLED, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert json.loads(rec.rows[0]["sampling"]) == {"seed": 42} async def test_absent_sampling_is_null(self): """无采样参数时为 NULL,而非空字符串或 "{}"——便于 SQL 过滤。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1608,6 +1719,7 @@ class TestEmitterSamplingColumn: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["sampling"] is None @@ -1631,7 +1743,7 @@ class TestEmitterCallerDimensions: async def test_every_entry_point_carries_the_dimensions(self, emit): """三条路径写出的行都必须带维度: 漏掉任一条,该租户的账就永远对不上。""" rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") if emit == "attempt": await emitter.emit_attempt( request=self._REQ_A, @@ -1641,12 +1753,19 @@ class TestEmitterCallerDimensions: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) elif emit == "cache_hit": - await emitter.emit_cache_hit(request=self._REQ_A, response=_resp(cache_hit=True)) + await emitter.emit_cache_hit( + request=self._REQ_A, response=_resp(cache_hit=True), operation="chat" + ) else: await emitter.emit_terminal_failure( - request=self._REQ_A, call_id="c", latency_ms=1, error="dead" + request=self._REQ_A, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) row = rec.rows[0] assert row["tenant_id"] == "tenant-a" @@ -1665,7 +1784,7 @@ class TestEmitterCallerDimensions: meta={"batch": "old-batch"}, ) rec = _MemoryRecorder() - mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None)) + mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None, scope="LLM")) async def terminal(request): # 缓存层回放的是历史那次的响应对象(其 call_id 属于 historical 那次) @@ -1687,7 +1806,7 @@ class TestEmitterCallerDimensions: JSON 函数直接查询,NULL 则要每条查询都额外判空。 """ rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_attempt( request=_REQ, # tenant_id=None, meta={} source=_source(), call_id="c", @@ -1695,6 +1814,7 @@ class TestEmitterCallerDimensions: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) row = rec.rows[0] assert row["tenant_id"] == "" @@ -1703,8 +1823,12 @@ class TestEmitterCallerDimensions: async def test_meta_is_serialized_with_sorted_keys(self): """键序固定,同一份维度在任意两行里字节一致,可直接做等值比对与去重。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=None).emit_terminal_failure( - request=self._REQ_A, call_id="c", latency_ms=1, error="dead" + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_terminal_failure( + request=self._REQ_A, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert list(json.loads(rec.rows[0]["meta"])) == ["a_first", "m_mid", "z_last"] @@ -1712,8 +1836,12 @@ class TestEmitterCallerDimensions: """`ensure_ascii=False`: 中文维度按原文落库,而非 `\\uXXXX` 转义串。""" rec = _MemoryRecorder() req = ChatRequest(messages=[{"role": "user", "content": "hi"}], meta={"dept": "研发"}) - await TelemetryEmitter(rec, text_cap=None).emit_terminal_failure( - request=req, call_id="c", latency_ms=1, error="dead" + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_terminal_failure( + request=req, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert "研发" in rec.rows[0]["meta"] @@ -1731,8 +1859,12 @@ class TestEmitterCallerDimensions: """ rec = _MemoryRecorder() req = ChatRequest(messages=[{"role": "user", "content": "hi"}], meta={"k": float("nan")}) - await TelemetryEmitter(rec, text_cap=None).emit_terminal_failure( - request=req, call_id="c", latency_ms=1, error="dead" + await TelemetryEmitter(rec, text_cap=None, scope="LLM").emit_terminal_failure( + request=req, + call_id="c", + error="dead", + operation="chat", + stats=_MIGRATED_STATS, ) assert rec.rows == [] @@ -1746,7 +1878,7 @@ class TestCostWithCachedTier: async def test_cached_hit_lowers_the_recorded_cost(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, pricing=self._TABLE, text_cap=None) + emitter = TelemetryEmitter(rec, pricing=self._TABLE, text_cap=None, scope="LLM") full = _resp(prompt_tokens=1_000_000, completion_tokens=0) await emitter.emit_attempt( request=_REQ, @@ -1756,6 +1888,7 @@ class TestCostWithCachedTier: response=full, error=None, reasoning_applies=True, + operation="chat", ) await emitter.emit_attempt( request=_REQ, @@ -1767,6 +1900,7 @@ class TestCostWithCachedTier: ), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] == pytest.approx(10.0) assert rec.rows[1]["cost"] == pytest.approx(5.2) # 400k×10 + 600k×2 @@ -1774,15 +1908,16 @@ class TestCostWithCachedTier: async def test_cache_hit_row_still_costs_zero(self): """缓存命中未产生新调用 → cost 恒 0.0,该短路必须排在任何换算之前。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, pricing=self._TABLE, text_cap=None).emit_cache_hit( + await TelemetryEmitter(rec, pricing=self._TABLE, text_cap=None, scope="LLM").emit_cache_hit( request=_REQ, response=_resp(prompt_tokens=1_000_000, cached_prompt_tokens=600_000), + operation="chat", ) assert rec.rows[0]["cost"] == 0.0 async def test_unavailable_usage_still_costs_none(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, pricing=self._TABLE, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, pricing=self._TABLE, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="c", @@ -1790,6 +1925,7 @@ class TestCostWithCachedTier: response=_resp(usage_source="unavailable", cached_prompt_tokens=5), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] is None @@ -1797,7 +1933,7 @@ class TestCostWithCachedTier: class TestEmitter: async def test_attempt_success_row(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(), @@ -1806,6 +1942,7 @@ class TestEmitter: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) row = rec.rows[0] assert row["call_id"] == "cid-1" and row["error"] is None @@ -1814,7 +1951,7 @@ class TestEmitter: async def test_attempt_failure_row(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(), @@ -1823,6 +1960,7 @@ class TestEmitter: response=None, error="TransientError: boom", reasoning_applies=True, + operation="chat", ) row = rec.rows[0] assert row["error"].startswith("TransientError") @@ -1832,8 +1970,14 @@ class TestEmitter: async def test_terminal_failure_row_is_unavailable(self): rec = _MemoryRecorder() - await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None).emit_terminal_failure( - request=_REQ, call_id="cid-t", latency_ms=5, error="cancelled" + await TelemetryEmitter( + rec, pricing=_PRICING, text_cap=None, scope="LLM" + ).emit_terminal_failure( + request=_REQ, + call_id="cid-t", + error="cancelled", + operation="chat", + stats=_MIGRATED_STATS, ) row = rec.rows[0] assert row["usage_source"] == "unavailable" and row["cost"] is None @@ -1845,7 +1989,7 @@ class TestEmitter: 参数第二组是改前兜底写出的 `0/4000` 形态: 那时换算出 0.032 的假金额。 """ rec = _MemoryRecorder() - await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="cid-u", @@ -1855,13 +1999,14 @@ class TestEmitter: ), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] is None async def test_measured_row_still_priced(self): """对照组: 同一价格表下 measured 行照常换算,证明 None 不是价格表没接上。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None).emit_attempt( + await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_source(), call_id="cid-m", @@ -1869,21 +2014,23 @@ class TestEmitter: response=_resp(prompt_tokens=0, completion_tokens=4000), error=None, reasoning_applies=True, + operation="chat", ) assert rec.rows[0]["cost"] == pytest.approx(0.032) async def test_cache_hit_keeps_zero_cost_even_when_unavailable(self): """缓存命中未产生新调用,0.0 是事实而非未知 → 短路必须排在 cache_hit 之后。""" rec = _MemoryRecorder() - await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None).emit_cache_hit( + await TelemetryEmitter(rec, pricing=_PRICING, text_cap=None, scope="LLM").emit_cache_hit( request=_REQ, response=_resp(cache_hit=True, usage_source="unavailable", completion_tokens=4000), + operation="chat", ) assert rec.rows[0]["cache_hit"] is True and rec.rows[0]["cost"] == 0.0 async def test_multimodal_messages_digested_before_storage(self): rec = _MemoryRecorder() - emitter = TelemetryEmitter(rec, text_cap=None) + emitter = TelemetryEmitter(rec, text_cap=None, scope="LLM") big = "data:image/png;base64," + "A" * 100_000 req = ChatRequest( messages=[ @@ -1903,6 +2050,7 @@ class TestEmitter: response=None, error="x", reasoning_applies=True, + operation="chat", ) assert len(rec.rows[0]["messages"]) < 500 # base64 不整段进库(VT R12) @@ -1911,7 +2059,7 @@ class TestEmitter: async def record_llm_call(self, **fields): raise OSError("disk full") - emitter = TelemetryEmitter(Broken(), text_cap=None) + emitter = TelemetryEmitter(Broken(), text_cap=None, scope="LLM") await emitter.emit_attempt( request=_REQ, source=_source(), @@ -1920,13 +2068,14 @@ class TestEmitter: response=_resp(), error=None, reasoning_applies=True, + operation="chat", ) # 不抛(降级不冒泡) class TestTelemetryMW: async def test_cache_hit_recorded(self): rec = _MemoryRecorder() - mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None)) + mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None, scope="LLM")) async def terminal(request): return _resp(cache_hit=True, latency_ms=0, call_id="cache-cid") @@ -1935,11 +2084,14 @@ class TestTelemetryMW: assert resp.cache_hit assert len(rec.rows) == 1 assert rec.rows[0]["cache_hit"] is True and rec.rows[0]["latency_ms"] == 0 + # 本层剩下的唯一职责就是这类行,故行形态在此钉死 + assert rec.rows[0]["event_kind"] == "cache_hit" + assert rec.rows[0]["operation"] == "chat" async def test_normal_success_not_double_recorded(self): """成功尝试由 RetryMW 逐次记录;最外层不得重复记。""" rec = _MemoryRecorder() - mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None)) + mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None, scope="LLM")) async def terminal(request): return _resp(cache_hit=False) @@ -1947,21 +2099,39 @@ class TestTelemetryMW: await mw(_REQ, terminal) assert rec.rows == [] - async def test_scope_level_failure_recorded(self): + async def test_scope_level_failure_is_not_written_here_anymore(self): + """1.3.5: 终态行改由 `GatewayClient.chat` 的公开边界写,本层不再写。 + + 两处同时写会让同一次失败出两条 `terminal_failure` 行,而下游正是按 + `WHERE event_kind = 'terminal_failure'` 计失败调用次数的——双写即双计。 + 异常本身仍须原样上抛。 + """ rec = _MemoryRecorder() - mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None)) + mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None, scope="LLM")) async def terminal(request): raise CircuitOpenError(scope="llm", retry_after_s=30.0) with pytest.raises(CircuitOpenError): await mw(_REQ, terminal) - assert len(rec.rows) == 1 and "circuit_open" in rec.rows[0]["error"] + assert rec.rows == [] + + async def test_cancellation_is_not_written_here_anymore(self): + """取消的终态行同样归公开边界;本层只负责不吞取消。""" + rec = _MemoryRecorder() + mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None, scope="LLM")) + + async def terminal(request): + raise asyncio.CancelledError + + with pytest.raises(asyncio.CancelledError): + await mw(_REQ, terminal) + assert rec.rows == [] async def test_attempt_level_failure_not_double_recorded(self): """RequestRejected 已被 RetryMW 逐次记录 → 最外层跳过。""" rec = _MemoryRecorder() - mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None)) + mw = TelemetryMW(TelemetryEmitter(rec, text_cap=None, scope="LLM")) async def terminal(request): raise RequestRejectedError("400") @@ -2009,7 +2179,7 @@ def _long_messages(): async def _emit_with_cap(messages, *, cap, response=_LONG, thinking=_LONG): rec = _MemoryRecorder() - await TelemetryEmitter(rec, text_cap=cap).emit_attempt( + await TelemetryEmitter(rec, text_cap=cap, scope="LLM").emit_attempt( request=ChatRequest(messages=messages, session_id="s"), source=_source(), call_id="c", @@ -2017,6 +2187,7 @@ async def _emit_with_cap(messages, *, cap, response=_LONG, thinking=_LONG): response=_resp(content=response, thinking=thinking), error=None, reasoning_applies=True, + operation="chat", ) return rec.rows[0] @@ -2091,7 +2262,7 @@ class TestTelemetryTextCap: """ for bad in (0, -1): with pytest.raises(ValueError, match="text_cap"): - TelemetryEmitter(_MemoryRecorder(), text_cap=bad) + TelemetryEmitter(_MemoryRecorder(), text_cap=bad, scope="LLM") class _StubEmbedTransport: @@ -2926,3 +3097,360 @@ class TestConcurrentReasoningPathContracts: assert attempts[0]["error"] finally: await client._transport.aclose() + + +# —— 1.3.5: 行级归因(scope/operation/event_kind)、诊断保真与终态行 —— + +_STATS = CallStats(logical_call_id="lcid-1", attempts=3, total_latency_ms=1234) + + +def _ctx_request(**overrides): + """带逻辑调用上下文的请求(库外部现场构造的请求恒无上下文)。""" + from polygateway.types import _CallContext + + base = {"messages": [{"role": "user", "content": "hi"}], "session_id": "sess-1"} + base.update(overrides) + return ChatRequest(call_context=_CallContext(now=lambda: 0.0), **base) + + +class TestRowLevelObservability: + """三类行的新列取值(1.3.5 设计 §5 表 + 实施计划 §3.3)。 + + 诊断列的存在理由是**归因**: 改前 `error` 是被 `str()` 压平的一列自由文本, + "哪个源回了什么状态码、网关正文说了什么"全部不可 SQL 化。 + """ + + def _emitter(self, rec, **kwargs): + return TelemetryEmitter(rec, scope="LLM", text_cap=None, **kwargs) + + async def _attempt_row(self, error, *, response=None, class_prefixed=False, request=_REQ): + rec = _MemoryRecorder() + await self._emitter(rec).emit_attempt( + request=request, + source=_source(), + call_id="c", + latency_ms=7, + response=response, + error=error, + reasoning_applies=True, + operation="chat", + class_prefixed_error=class_prefixed, + ) + return rec.rows[0] + + async def test_attempt_row_carries_scope_operation_and_kind(self): + row = await self._attempt_row(None, response=_resp()) + assert row["scope"] == "LLM" # 构造期注入,不拿 source_name 顶替 + assert row["operation"] == "chat" + assert row["event_kind"] == "attempt" + + async def test_success_row_leaves_every_diagnostic_null(self): + """成功行不统一填 200: 那会让"有状态码"不再等价于"这次失败了"。""" + row = await self._attempt_row(None, response=_resp()) + assert row["http_status_code"] is None + assert row["error_type"] is None and row["cause_type"] is None + assert row["error_body"] is None + # attempts / total_latency_ms 只属终态行 + assert row["attempts"] is None and row["total_latency_ms"] is None + + async def test_relabelled_status_is_recorded_as_received(self): + """中转把 529 改写成 503 → 如实记 503,**不猜回 529**(设计 §1)。""" + exc = TransientError( + "s1 瞬时错误: 503 | upstream said 529", + source_name="s1", + status_code=503, + operation="chat", + body_text="upstream said 529", + ) + exc.__cause__ = httpx.ConnectTimeout("") + row = await self._attempt_row(exc) + assert row["http_status_code"] == 503 + assert row["error_type"] == "TransientError" + assert row["cause_type"] == "ConnectTimeout" + assert row["error_body"] == "upstream said 529" + assert row["error"] == "s1 瞬时错误: 503 | upstream said 529" + + async def test_direct_529_is_recorded_as_529(self): + exc = TransientError("s1 瞬时错误: 529", status_code=529, body_text="overloaded") + row = await self._attempt_row(exc) + assert row["http_status_code"] == 529 + + @pytest.mark.parametrize( + "cause", + [ + httpx.ConnectTimeout(""), + httpx.ReadTimeout(""), + httpx.WriteTimeout(""), + httpx.PoolTimeout(""), + ], + ) + async def test_empty_timeout_text_still_yields_a_cause_type(self, cause): + """`str()` 为空的 httpx 超时: 类型必须留在 `cause_type` 里,否则无从分辨。""" + exc = TransientError("") + exc.__cause__ = cause + row = await self._attempt_row(exc) + assert row["cause_type"] == type(cause).__name__ + assert row["error"] == "TransientError" # 空 str() 退回类名,不落空串 + + async def test_error_body_is_the_bounded_summary_not_the_model_output(self): + """`error_body` 是**网关拒绝时说的话**(`summarize_body` 摘要),不是模型正文。""" + exc = ResultInvalidError("坏结果", raw_text="x" * 5000, body_text="") + row = await self._attempt_row(exc) + assert row["error_body"] is None # 未知一律 None,不拿 raw_text 顶替 + assert "x" * 100 not in (row["error"] or "") + + async def test_string_error_is_never_parsed_for_diagnostics(self): + """取消行沿用既有 `"cancelled"`;字符串不解析猜诊断(设计 §5)。""" + row = await self._attempt_row("cancelled") + assert row["error"] == "cancelled" + assert row["error_type"] is None + assert row["cause_type"] is None and row["http_status_code"] is None + assert row["error_body"] is None + + async def test_class_prefix_is_an_explicit_policy_parameter(self): + """OCR 的 `"类名: msg"` 口径由出口显式参数保留,不在三处复制拼装。""" + exc = RequestRejectedError("m1 请求被拒: 400") + assert (await self._attempt_row(exc))["error"] == "m1 请求被拒: 400" + prefixed = await self._attempt_row(exc, class_prefixed=True) + assert prefixed["error"] == "RequestRejectedError: m1 请求被拒: 400" + + async def test_logical_call_id_comes_from_the_context(self): + request = _ctx_request() + row = await self._attempt_row(None, response=_resp(), request=request) + assert row["logical_call_id"] == request.call_context.logical_call_id + + async def test_absent_context_lands_null_instead_of_a_fabricated_id(self): + """库内现场构造的请求没有上下文 → NULL,**不造 ID**(设计 §5 I5)。""" + row = await self._attempt_row(None, response=_resp()) + assert row["logical_call_id"] is None + + async def test_cache_hit_row_kind_and_null_logical_counters(self): + rec = _MemoryRecorder() + await self._emitter(rec).emit_cache_hit( + request=_REQ, response=_resp(cache_hit=True), operation="chat" + ) + row = rec.rows[0] + assert row["event_kind"] == "cache_hit" and row["scope"] == "LLM" + assert row["operation"] == "chat" + assert row["attempts"] is None and row["total_latency_ms"] is None + assert row["error_type"] is None and row["http_status_code"] is None + + async def test_terminal_row_carries_the_frozen_snapshot(self): + rec = _MemoryRecorder() + await self._emitter(rec).emit_terminal_failure( + request=_REQ, + call_id="c", + error=AllSourcesExhausted(scope="llm", reason="retry_exhausted", retry_after_s=2.0), + operation="chat", + stats=_STATS, + ) + row = rec.rows[0] + assert row["event_kind"] == "terminal_failure" + assert row["attempts"] == 3 + # 两列同取一份冻结快照,避免双时钟微差 + assert row["total_latency_ms"] == 1234 and row["latency_ms"] == 1234 + assert row["error_type"] == "AllSourcesExhausted" + assert row["error"] == "llm 网关暂时不可用: retry_exhausted" + + async def test_terminal_row_never_borrows_the_last_attempt_diagnostics(self): + """C1 红线: 终态三列保持 NULL——不拿最后一个源的现场冒充整池归因。""" + exc = AllSourcesExhausted(scope="llm", reason="retry_exhausted", retry_after_s=2.0) + exc.__cause__ = TransientError("s1 503", status_code=503, body_text="gateway said 529") + rec = _MemoryRecorder() + await self._emitter(rec).emit_terminal_failure( + request=_REQ, call_id="c", error=exc, operation="chat", stats=_STATS + ) + row = rec.rows[0] + assert row["http_status_code"] is None + assert row["cause_type"] is None + assert row["error_body"] is None + + async def test_terminal_row_costs_nothing(self): + """费用聚合口径不变: 终态行 cost 恒 NULL、usage `unavailable`、token 0。""" + rec = _MemoryRecorder() + await TelemetryEmitter( + rec, scope="LLM", pricing=_PRICING, text_cap=None + ).emit_terminal_failure( + request=_REQ, call_id="c", error="cancelled", operation="chat", stats=_STATS + ) + row = rec.rows[0] + assert row["cost"] is None and row["usage_source"] == "unavailable" + assert row["prompt_tokens"] == 0 and row["completion_tokens"] == 0 + + async def test_structured_exhaustion_terminal_explains_itself_within_bounds(self): + """C2: 结构化耗尽的终态是唯一记录,故有界说明并入 `error`,且不含 raw_text。""" + exc = ResultInvalidError( + "结构化输出阶梯耗尽", + raw_text="y" * 5000, + repair_error="r" * 500, + validation_errors=tuple("v" * 500 for _ in range(5)), + ) + rec = _MemoryRecorder() + await self._emitter(rec).emit_terminal_failure( + request=_REQ, call_id="c", error=exc, operation="chat", stats=_STATS + ) + error = rec.rows[0]["error"] + assert error.startswith("结构化输出阶梯耗尽") + assert "repair=" in error and "validation=" in error + assert "y" * 50 not in error # 模型正文预算已由 attempt 行承担,不重复落库 + # 至多 3 条 × 200 字符 + repair 200 字符,整体有界 + assert error.count("v" * 200) == 3 + assert len(error) < 1200 + + @pytest.mark.parametrize("operation", ["chat", "embed", "recognize_text", "parse_layout"]) + async def test_operation_is_given_by_the_call_site_not_the_exception(self, operation): + """新列 `operation` 与 `exc.operation` 是两个语义(设计 §5 I1/I2)。""" + exc = TransientError("boom", status_code=500, operation="download_result") + rec = _MemoryRecorder() + await self._emitter(rec).emit_attempt( + request=_REQ, + source=_source(), + call_id="c", + latency_ms=1, + response=None, + error=exc, + reasoning_applies=False, + operation=operation, + ) + assert rec.rows[0]["operation"] == operation + + +class TestRecorderShapeGate: + """C3 装配闸: 旧签名 recorder 必须在装配期报错,而不是运行期静默丢行。 + + `_record` 的 `except Exception` 会把旧 recorder 的 `TypeError` 吞成 warning, + 后果是自定义 recorder 在下游升级后 100% 丢遥测且调用照常成功。 + """ + + def test_old_signature_recorder_is_refused_at_assembly(self): + class Old: + async def record_llm_call(self, *, call_id, model, error) -> None: ... + + with pytest.raises(ValueError, match="record_llm_call"): + TelemetryEmitter(Old(), scope="LLM", text_cap=None) + + def test_kwargs_recorder_passes(self): + TelemetryEmitter(_MemoryRecorder(), scope="LLM", text_cap=None) # 不抛 + + def test_uninspectable_recorder_is_a_configuration_error(self): + """不可 inspect(C 实现等)按配置错误当场报错,不进入运行期静默丢行。""" + + class Opaque: + record_llm_call = print # 内置函数: inspect.signature 拿不到 + + with pytest.raises(ValueError): + TelemetryEmitter(Opaque(), scope="LLM", text_cap=None) + + def test_gate_derives_parameter_names_from_the_protocol(self, monkeypatch): + """参数名从协议签名派生,不手抄第四份清单——改协议,闸自动跟随。""" + import polygateway.middleware.telemetry as tele + + class NarrowProtocol: + async def record_llm_call(self, *, call_id, model) -> None: ... + + monkeypatch.setattr(tele, "TelemetryRecorder", NarrowProtocol) + + class NarrowRecorder: + async def record_llm_call(self, *, call_id, model) -> None: ... + + # 若闸内硬编码了 36 个名字,这里必然拒绝 + tele.TelemetryEmitter(NarrowRecorder(), scope="LLM", text_cap=None) + + +class TestTerminalRowSqlSemantics: + """§5 的归因查询与 §8 的迁移清单,按**真实 SQLite 落库**断言。 + + 这些断言的对象是下游真正会写的 SQL: 改前"计失败调用次数"只能按 + `error IS NOT NULL`,而那会同时命中尝试行与终态行。 + """ + + async def _failing_chat_db(self, tmp_path): + from tests.unit.test_client import _client + + recorder = SQLiteRecorder(tmp_path / "t.db", auto_migrate=True) + client = _client( + handler=lambda request: httpx.Response(503), + telemetry=recorder, + retry=RetryPolicy(2, 0.001, 0.01), + ) + try: + with pytest.raises(AllSourcesExhausted): + await client.chat([{"role": "user", "content": "hi"}]) + finally: + await client._transport.aclose() + recorder.close() + return sqlite3.connect(tmp_path / "t.db") + + async def test_attribution_query_gives_terminal_reason_and_per_source_scene(self, tmp_path): + """设计 §5 验收查询: 一条 `logical_call_id` 同时给出整池终态与逐源现场。""" + conn = await self._failing_chat_db(tmp_path) + lcid = conn.execute( + "SELECT logical_call_id FROM llm_calls WHERE event_kind='terminal_failure'" + ).fetchone()[0] + assert lcid is not None + rows = conn.execute( + "SELECT event_kind, source_name, http_status_code, error_type, error " + "FROM llm_calls WHERE logical_call_id = ? ORDER BY created_at", + (lcid,), + ).fetchall() + kinds = [r[0] for r in rows] + assert kinds.count("attempt") == 2 # 逐源现场 + assert kinds.count("terminal_failure") == 1 # 整池终态 + attempts = [r for r in rows if r[0] == "attempt"] + assert all(a[2] == 503 for a in attempts) # 每个源怎么死的 + terminal = next(r for r in rows if r[0] == "terminal_failure") + assert terminal[3] == "AllSourcesExhausted" + assert "retry_exhausted" in terminal[4] # 整池为何失败 + assert terminal[2] is None # C1: 终态不冒充逐源状态码 + + async def test_failure_count_must_come_from_terminal_rows_only(self, tmp_path): + """`error IS NOT NULL` 跨两类行,不再是"失败调用数"的判据(§8)。""" + conn = await self._failing_chat_db(tmp_path) + (by_error,) = conn.execute( + "SELECT COUNT(*) FROM llm_calls WHERE error IS NOT NULL" + ).fetchone() + (by_terminal,) = conn.execute( + "SELECT COUNT(*) FROM llm_calls WHERE event_kind = 'terminal_failure'" + ).fetchone() + assert by_error == 3 # 2 条尝试错误行 + 1 条终态行 + assert by_terminal == 1 # 一次逻辑调用 = 一次失败 + + async def test_terminal_rows_never_contribute_to_cost(self, tmp_path): + """费用聚合口径不变: 终态行 cost 恒 NULL、usage `unavailable`、token 0。""" + conn = await self._failing_chat_db(tmp_path) + (billable,) = conn.execute( + "SELECT COUNT(*) FROM llm_calls " + "WHERE event_kind = 'terminal_failure' AND cost IS NOT NULL" + ).fetchone() + assert billable == 0 + row = conn.execute( + "SELECT usage_source, prompt_tokens, completion_tokens FROM llm_calls " + "WHERE event_kind = 'terminal_failure'" + ).fetchone() + assert row == ("unavailable", 0, 0) + + async def test_latency_must_be_grouped_by_event_kind(self, tmp_path): + """终态行携带**逻辑总耗时**,量级大于单次尝试 → 时延看板必须分组(§8)。""" + conn = await self._failing_chat_db(tmp_path) + grouped = dict( + conn.execute( + "SELECT event_kind, MAX(latency_ms) FROM llm_calls GROUP BY event_kind" + ).fetchall() + ) + assert set(grouped) == {"attempt", "terminal_failure"} + # 终态是整次调用的耗时,含退避与两次尝试,故不小于任何单次尝试 + assert grouped["terminal_failure"] >= grouped["attempt"] + terminal = conn.execute( + "SELECT latency_ms, total_latency_ms, attempts FROM llm_calls " + "WHERE event_kind = 'terminal_failure'" + ).fetchone() + assert terminal[0] == terminal[1] # 同一份冻结快照,无双时钟微差 + assert terminal[2] == 2 + # 尝试行的逻辑两列恒 NULL + assert ( + conn.execute( + "SELECT COUNT(*) FROM llm_calls WHERE event_kind = 'attempt' " + "AND (attempts IS NOT NULL OR total_latency_ms IS NOT NULL)" + ).fetchone()[0] + == 0 + ) diff --git a/tests/unit/test_usage_source_domain.py b/tests/unit/test_usage_source_domain.py index 8c583a5..36fff7a 100644 --- a/tests/unit/test_usage_source_domain.py +++ b/tests/unit/test_usage_source_domain.py @@ -35,6 +35,7 @@ from polygateway.types import ( USAGE_SOURCES, BackpressurePolicy, BreakerConfig, + CallStats, ChatRequest, EmbeddingTransportResult, GlobalLimits, @@ -44,6 +45,9 @@ from polygateway.types import ( SourceConfig, ) +# 终态行的快照入参(1.3.5): `emit_terminal_failure` 不再收 `latency_ms`。 +_MIGRATED_STATS = CallStats(logical_call_id="lcid-mig", attempts=1, total_latency_ms=1) + _REQ = ChatRequest(messages=[{"role": "user", "content": "hi"}]) _DOMAIN = sorted(USAGE_SOURCES) @@ -255,7 +259,7 @@ def _resp(usage_source): @pytest.mark.parametrize("emitted", _DOMAIN) async def test_emit_attempt_success_stays_in_domain(emitted): recorder = _MemoryRecorder() - await TelemetryEmitter(recorder, text_cap=None).emit_attempt( + await TelemetryEmitter(recorder, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_src(), call_id="cid", @@ -263,6 +267,7 @@ async def test_emit_attempt_success_stays_in_domain(emitted): response=_resp(emitted), error=None, reasoning_applies=True, + operation="chat", ) assert recorder.rows[0]["usage_source"] in USAGE_SOURCES @@ -270,7 +275,7 @@ async def test_emit_attempt_success_stays_in_domain(emitted): async def test_emit_attempt_failed_attempt_stays_in_domain(): """失败尝试无 response,`usage_source` 取 emitter 自己的字面量。""" recorder = _MemoryRecorder() - await TelemetryEmitter(recorder, text_cap=None).emit_attempt( + await TelemetryEmitter(recorder, text_cap=None, scope="LLM").emit_attempt( request=_REQ, source=_src(), call_id="cid", @@ -278,6 +283,7 @@ async def test_emit_attempt_failed_attempt_stays_in_domain(): response=None, error="boom", reasoning_applies=True, + operation="chat", ) assert recorder.rows[0]["usage_source"] in USAGE_SOURCES @@ -285,8 +291,10 @@ async def test_emit_attempt_failed_attempt_stays_in_domain(): @pytest.mark.parametrize("emitted", _DOMAIN) async def test_emit_cache_hit_stays_in_domain(emitted): recorder = _MemoryRecorder() - await TelemetryEmitter(recorder, text_cap=None).emit_cache_hit( - request=_REQ, response=_resp(emitted) + await TelemetryEmitter(recorder, text_cap=None, scope="LLM").emit_cache_hit( + request=_REQ, + response=_resp(emitted), + operation="chat", ) assert recorder.rows[0]["usage_source"] in USAGE_SOURCES @@ -294,7 +302,11 @@ async def test_emit_cache_hit_stays_in_domain(emitted): async def test_emit_terminal_failure_stays_in_domain(): """终态失败无具体源,`usage_source` 同样取 emitter 字面量。""" recorder = _MemoryRecorder() - await TelemetryEmitter(recorder, text_cap=None).emit_terminal_failure( - request=_REQ, call_id="cid", latency_ms=10, error="cancelled" + await TelemetryEmitter(recorder, text_cap=None, scope="LLM").emit_terminal_failure( + request=_REQ, + call_id="cid", + error="cancelled", + operation="chat", + stats=_MIGRATED_STATS, ) assert recorder.rows[0]["usage_source"] in USAGE_SOURCES From 7b2f6105f340abe58a5833fd04f049947ac09814 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 12:17:34 -0400 Subject: [PATCH 07/10] test: cover the ten call observability columns on real Postgres Migrate the PG telemetry fixtures to the 36-field recorder and add the storage compatibility acceptance the plan calls for. Mechanical migration: - _EXPECTED_COLUMNS 27 -> 37 physical columns - _record_minimal gains the ten keys in the same shape as the unit suite - _PRE_TENANT_COLUMNS now excludes 14 columns, derived from _CALL_OBSERVABILITY_COLUMNS instead of a second hand-written list, and the two manual-mode warnings assert a notice derived from COLUMNS order so a column that silently drops out of the warning turns the test red New TestCallObservabilityColumnsAcceptance, all on a 27-column 1.3.4 shaped table built by the existing pg_sandbox factory: - auto appends the ten columns in the same order as a fresh database and old rows keep NULL in every one of them (no backfill, no sentinel) - manual sends no DDL, trims the INSERT, and still round-trips the other 26 columns value by value - an old-version writer using insert_sql with the 1.3.4 column set and a new-version writer share one table, and event_kind filtering counts neither the old rows as failures nor as successes _minimal_fields is split out of _record_minimal so the simulated old process reuses the same values rather than copying them. Verified against the real lab Postgres: 30 passed. --- tests/integration/test_postgres_telemetry.py | 389 +++++++++++++++++-- 1 file changed, 363 insertions(+), 26 deletions(-) diff --git a/tests/integration/test_postgres_telemetry.py b/tests/integration/test_postgres_telemetry.py index 380e223..3e5055b 100644 --- a/tests/integration/test_postgres_telemetry.py +++ b/tests/integration/test_postgres_telemetry.py @@ -28,7 +28,7 @@ import pytest from dotenv import dotenv_values from polygateway.telemetry.postgres import PostgresRecorder -from polygateway.telemetry.schema import COLUMNS, telemetry_schema_sql +from polygateway.telemetry.schema import COLUMNS, insert_sql, telemetry_schema_sql _EXPECTED_COLUMNS = [ "call_id", @@ -58,8 +58,31 @@ _EXPECTED_COLUMNS = [ "meta", "thinking_observation", "reasoning_effort", + # —— 1.3.5 逻辑调用统计与结构化失败诊断的十列(issue #19/#23)—— + "scope", + "operation", + "logical_call_id", + "event_kind", + "http_status_code", + "error_type", + "cause_type", + "error_body", + "attempts", + "total_latency_ms", ] +# 1.3.5 之前那张表的 27 个物理列(26 个 INSERT 字段 + created_at)。写成固定切片 +# 而非 `[:-10]`: 后者会随下一次补列静默漂移到另一张表上,而漂移的表现是 +# "旧表补列"用例悄悄改测了别的形态。 +_PRE_135_COLUMNS = _EXPECTED_COLUMNS[:27] + +# 1.3.4 版本的 recorder 实际写入的列(物理列去掉库从不显式写的 created_at) +_PRE_135_WRITTEN_COLUMNS = [c for c in _PRE_135_COLUMNS if c != "created_at"] + +# 1.3.5 新增的十列,按 `COLUMNS`(即 INSERT 字段序)排列: manual 档告警逐字比对与 +# "旧行新列为 NULL"两处共用同一份,免得两边各抄一份后各自漂移。 +_CALL_OBSERVABILITY_COLUMNS = [c for c in COLUMNS if c not in _PRE_135_WRITTEN_COLUMNS] + def _dsn() -> str | None: """读 `.env` 的 DSN 并剥掉 SQLAlchemy 风格的 `+driver` 后缀;未配置返回 None。""" @@ -92,13 +115,13 @@ async def template_admin_dsn() -> str: return value -async def _record_minimal( - recorder: PostgresRecorder, call_id: str | None = None, **overrides -) -> dict[str, object]: - """记一行最小遥测,并**返回实际提交的字段**供调用方逐列比对回读结果。 +def _minimal_fields(call_id: str | None = None, **overrides) -> dict[str, object]: + """一行最小遥测的**完整字段字典**(不写库),供 recorder 写入与旧版本进程模拟共用。 - 返回值不是顺手加的: 逐列断言若在测试里另抄一份期望值,抄错的那一列会以 - "库写错列位"的形态误报,而漏抄的列则悄悄不被验证。 + 独立出来不是顺手拆的: "新旧进程混写"那条用例要以 1.3.4 的列集直接发 INSERT, + 若它另抄一份取值,抄错的那一列会以"库写错列位"的形态误报。 + + 调用方随后逐列比对回读结果,故返回的就是实际提交的那一份。 """ fields: dict[str, object] = { "call_id": call_id if call_id is not None else "c1", @@ -130,8 +153,30 @@ async def _record_minimal( "thinking_observation": "unknown", # 同理: `Effort` 归一成裸 str,不表态则是 None(与 'low' 必须分得开) "reasoning_effort": None, + # —— 1.3.5 十列: 默认形态即"一次普通尝试行"(与单测 `_record_minimal` 同款)—— + "scope": "LLM", + "operation": "chat", + # 库内现场构造的请求没有上下文 → NULL,不造 ID + "logical_call_id": None, + "event_kind": "attempt", + # 诊断四列只在失败的 attempt 行上非空;成功行不统一填 200 + "http_status_code": None, + "error_type": None, + "cause_type": None, + "error_body": None, + # 逻辑快照两列只属终态行 + "attempts": None, + "total_latency_ms": None, } fields.update(overrides) + return fields + + +async def _record_minimal( + recorder: PostgresRecorder, call_id: str | None = None, **overrides +) -> dict[str, object]: + """记一行最小遥测,并**返回实际提交的字段**供调用方逐列比对回读结果。""" + fields = _minimal_fields(call_id, **overrides) await recorder.record_llm_call(**fields) return fields @@ -167,6 +212,17 @@ async def _fetch(dsn: str, sql: str, *args): await conn.close() +async def _execute_args(dsn: str, sql: str, *args) -> None: + """带参数执行单条语句(扩展协议);用于模拟旧版本进程按旧列集发出的 INSERT。""" + import asyncpg + + conn = await asyncpg.connect(dsn, timeout=10) + try: + await conn.execute(sql, *args) + finally: + await conn.close() + + async def _execute_script(dsn: str, sql: str) -> None: """整段执行多语句脚本(不带参数,走简单查询协议)——模拟下游把脚本贴进 psql。""" import asyncpg @@ -588,14 +644,24 @@ _PRE_TENANT_INSERT = ( ) -# `_PRE_TENANT_DDL` 的物理列(23 个): 由 `_EXPECTED_COLUMNS` 去掉此后新增的四列 +# `_PRE_TENANT_DDL` 的物理列(23 个): 由 `_EXPECTED_COLUMNS` 去掉此后新增的列 # 派生而非另抄一份——两份常量必然漂移,而漂移的表现是"manual 档没补列"这条断言假绿。 -# 去掉后的顺序与 DDL 逐字一致(这四列在 DDL 里本就排在末尾)。 -_PRE_TENANT_COLUMNS = [ - c - for c in _EXPECTED_COLUMNS - if c not in ("tenant_id", "meta", "thinking_observation", "reasoning_effort") -] +# 去掉后的顺序与 DDL 逐字一致(这些列在 DDL 里本就排在末尾)。 +_PRE_TENANT_ABSENT = ( + "tenant_id", + "meta", + "thinking_observation", + "reasoning_effort", + *_CALL_OBSERVABILITY_COLUMNS, +) + +_PRE_TENANT_COLUMNS = [c for c in _EXPECTED_COLUMNS if c not in _PRE_TENANT_ABSENT] + +# 这张表缺的 14 个维度,按 `COLUMNS`(即告警的排列序)列出: manual 档告警逐字比对用。 +# 逐字而非前缀断言,是为了让"将来漏进告警的新列"当场红(设计 §4.2 的纪律)。 +_PRE_TENANT_MISSING_NOTICE = ( + f"以下维度不会被记录: {', '.join(c for c in COLUMNS if c in _PRE_TENANT_ABSENT)}。" +) # 回读要逐列比对的字段: 物理列去掉库从不显式写的 created_at,恰好 22 个 _PRE_TENANT_WRITTEN_COLUMNS = [c for c in _PRE_TENANT_COLUMNS if c != "created_at"] @@ -703,7 +769,7 @@ class TestCallerDimensionsAcceptance: "WHERE table_schema = $1 AND table_name = 'llm_calls' ORDER BY ordinal_position", schema, ) - # 22 → 26 个 recorder 字段(加 created_at 共 27 个物理列),且新列追加在末尾 + # 22 → 36 个 recorder 字段(加 created_at 共 37 个物理列),且新列追加在末尾 assert [r["column_name"] for r in cols] == _EXPECTED_COLUMNS rows = await _fetch( schema_dsn, @@ -908,11 +974,8 @@ class TestManualSchemaModeAcceptance: assert [m for m in captured_warnings if "补列失败" in m] == [] notices = [m for m in captured_warnings if "auto_migrate=False" in m] assert len(notices) == 1 # 准备期一次讲清,不逐行刷屏 - # 逐字钉住四个维度: 前缀断言会让将来漏进告警的新列照样绿 - assert ( - "以下维度不会被记录: tenant_id, meta, thinking_observation, reasoning_effort。" - in notices[0] - ) + # 逐字钉住缺的每一个维度: 前缀断言会让将来漏进告警的新列照样绿 + assert _PRE_TENANT_MISSING_NOTICE in notices[0] finally: await recorder.aclose() @@ -938,11 +1001,8 @@ class TestManualSchemaModeAcceptance: assert recorder.telemetry_status.degraded is False notices = [m for m in captured_warnings if "auto_migrate=False" in m] assert len(notices) == 1 # 准备期一次,第二行不再重复 - # 逐字钉住四个维度: 前缀断言会让将来漏进告警的新列照样绿 - assert ( - "以下维度不会被记录: tenant_id, meta, thinking_observation, reasoning_effort。" - in notices[0] - ) + # 逐字钉住缺的每一个维度: 前缀断言会让将来漏进告警的新列照样绿 + assert _PRE_TENANT_MISSING_NOTICE in notices[0] # 提示里的 SQL 必须可直接粘贴执行,而不是只报个列名 assert ( "ALTER TABLE llm_calls ADD COLUMN tenant_id TEXT NOT NULL DEFAULT '';" in notices[0] @@ -974,6 +1034,283 @@ class TestManualSchemaModeAcceptance: await recorder.aclose() +# 1.3.5 之前(1.3.4 发布形态)的表: 26 个 recorder 字段 + created_at = 27 个物理列, +# 没有本版新增的任何一列。裸表名: 由 `pg_sandbox` 在沙箱 schema 的 search_path 下执行。 +_PRE_135_DDL = """ +CREATE TABLE llm_calls ( + call_id TEXT PRIMARY KEY, + parent_call_id TEXT, + session_id TEXT, + model TEXT NOT NULL, + provider TEXT NOT NULL, + source_name TEXT NOT NULL, + messages TEXT NOT NULL, + response TEXT NOT NULL, + thinking TEXT NOT NULL DEFAULT '', + prompt_tokens INTEGER NOT NULL, + completion_tokens INTEGER NOT NULL, + usage_source TEXT NOT NULL, + latency_ms INTEGER NOT NULL, + ttft_ms DOUBLE PRECISION, + max_inter_token_ms DOUBLE PRECISION, + cache_hit BOOLEAN NOT NULL DEFAULT FALSE, + error TEXT, + cost DOUBLE PRECISION, + created_at TIMESTAMPTZ NOT NULL DEFAULT now(), + cached_prompt_tokens INTEGER, + model_reported TEXT, + sampling TEXT, + reasoning_tokens INTEGER, + tenant_id TEXT NOT NULL DEFAULT '', + meta JSONB NOT NULL DEFAULT '{}'::jsonb, + thinking_observation TEXT, + reasoning_effort TEXT +) +""" + +# 一行 1.3.5 之前写下的历史数据(只列 NOT NULL 列,与当年 recorder 的写入等价)。 +# 工厂的 `extra` 逐条裸执行、不接受查询参数,故 call_id 内联成字面量。 +_PRE_135_INSERT = ( + "INSERT INTO llm_calls (call_id, model, provider, source_name, messages, response, " + "prompt_tokens, completion_tokens, usage_source, latency_ms) " + "VALUES ('pre135', 'm', 'p', 's1', '[]', 'old body', 1, 2, 'measured', 10)" +) + +# 这张表缺的正是本版十列;manual 档告警要逐字比对的那句。 +_PRE_135_MISSING_NOTICE = f"以下维度不会被记录: {', '.join(_CALL_OBSERVABILITY_COLUMNS)}。" + + +@pytest.fixture +async def pre_135_schema(pg_sandbox) -> tuple[str, str]: + """一次性沙箱里造一张 **1.3.4 形态的 27 列表**,并留一行本版之前的历史数据。 + + 共享表 `llm_calls` 一个字节都不碰: 本机那张表升级一次就再也回不到旧形态, + 指望它还是旧形态的测试第二次跑就会空转(与 `pre_tenant_schema` 同款理由)。 + """ + sandbox = await pg_sandbox(ddl=_PRE_135_DDL, extra=(_PRE_135_INSERT,)) + return sandbox.dsn, sandbox.schema + + +class TestCallObservabilityColumnsAcceptance: + """1.3.5(issue #19/#23)的 PG 存储兼容验收: auto 追加 / manual 裁剪 / 旧行 NULL / 混写。 + + 单元层在 SQLite 上断的是同一族语义,这里断的是**真实 PG 上确实如此**—— + 两端的 DDL、补列语句与列序是两份文本(`SQLITE_BACKFILL` 与 `_PG_BACKFILL_DECLS`), + 只有真表能证明它们没有分叉。 + """ + + async def test_pre_135_table_gains_the_ten_columns_and_old_rows_stay_null(self, pre_135_schema): + """27 列旧表 auto 补齐到 37 列,新行两类取值读得回,**历史行十列一律 NULL**。 + + 旧行不回填是本版的明示决策(设计 §5): NULL 表达的是"补列之前根本没记过 + 这件事",与任何哨兵值都不是一回事。若哪天有人给这十列加了 DEFAULT,历史行 + 会被就地改写成"记过且值为 X",归因查询从此分不清真实缺口——故这条断言的 + 方向是 NULL,不是空串也不是 0。 + """ + schema_dsn, schema = pre_135_schema + recorder = _recorder(schema_dsn, auto_migrate=True) + try: + # attempt 行: 诊断四列非空、逻辑快照两列 NULL + await _record_minimal( + recorder, + call_id="att", + scope="LLM", + operation="chat", + logical_call_id="lcid-1", + event_kind="attempt", + http_status_code=503, + error_type="TransientError", + cause_type="ReadTimeout", + error_body="upstream said 503", + ) + # 终态行: 逻辑快照两列非空、诊断三列 NULL(不搬最后一次 attempt 的现场) + await _record_minimal( + recorder, + call_id="term", + scope="LLM", + operation="chat", + logical_call_id="lcid-1", + event_kind="terminal_failure", + error="AllSourcesExhausted: 全部源已耗尽", + error_type="AllSourcesExhausted", + attempts=3, + total_latency_ms=4200, + ) + cols = await _fetch( + schema_dsn, + "SELECT column_name FROM information_schema.columns " + "WHERE table_schema = $1 AND table_name = 'llm_calls' ORDER BY ordinal_position", + schema, + ) + # 26 → 36 个 recorder 字段(加 created_at 共 37 个物理列),新列追加在末尾: + # 列序与新建库一致才不会让两条升级路径分叉 + assert [r["column_name"] for r in cols] == _EXPECTED_COLUMNS + + names = ", ".join(_CALL_OBSERVABILITY_COLUMNS) + rows = await _fetch( + schema_dsn, + f"SELECT call_id, {names} FROM llm_calls " + "WHERE call_id = ANY($1::text[]) ORDER BY call_id", + ["att", "pre135", "term"], + ) + by_id = {r["call_id"]: r for r in rows} + + assert dict(by_id["att"]) == { + "call_id": "att", + "scope": "LLM", + "operation": "chat", + "logical_call_id": "lcid-1", + "event_kind": "attempt", + "http_status_code": 503, + "error_type": "TransientError", + "cause_type": "ReadTimeout", + "error_body": "upstream said 503", + "attempts": None, + "total_latency_ms": None, + } + assert dict(by_id["term"]) == { + "call_id": "term", + "scope": "LLM", + "operation": "chat", + "logical_call_id": "lcid-1", + "event_kind": "terminal_failure", + "http_status_code": None, + "error_type": "AllSourcesExhausted", + "cause_type": None, + "error_body": None, + "attempts": 3, + "total_latency_ms": 4200, + } + # 历史行: 十列逐列 NULL(整体比对,漏掉任一列都红) + assert dict(by_id["pre135"]) == { + "call_id": "pre135", + **dict.fromkeys(_CALL_OBSERVABILITY_COLUMNS), + } + finally: + await recorder.aclose() + + async def test_manual_trims_the_insert_on_a_pre_135_table( + self, pre_135_schema, captured_warnings + ): + """27 列旧表 + manual: 一条 DDL 都不发,写入按现有列裁剪后照样落库。 + + 与上一条恰成对照: 同一张表、同一份负载,只有 `auto_migrate` 不同,列数就必须是 + 27 与 37 之别。裁剪是关掉 ALTER 的前提——不裁剪的话每行 INSERT 都撞缺列 + (SQLSTATE 42703)而被整行丢弃,那是把自动补列换成遥测静默全失。 + """ + schema_dsn, schema = pre_135_schema + recorder = _recorder(schema_dsn, auto_migrate=False) + try: + recorded = await _record_minimal( + recorder, call_id="man135", scope="LLM", logical_call_id="lcid-x" + ) + cols = await _fetch( + schema_dsn, + "SELECT column_name FROM information_schema.columns " + "WHERE table_schema = $1 AND table_name = 'llm_calls' ORDER BY ordinal_position", + schema, + ) + # 表结构逐字不动: 既没多出十列,也没被顺手改了列序 + assert [r["column_name"] for r in cols] == _PRE_135_COLUMNS + + names = ", ".join(_PRE_135_WRITTEN_COLUMNS) + rows = await _fetch( + schema_dsn, f"SELECT {names} FROM llm_calls WHERE call_id = $1", "man135" + ) + assert len(rows) == 1 # 裁剪后的 INSERT 真写进去了,不是被 PG 拒收 + # 其余 26 列逐列与提交值相等: 少写十列最容易引发的错是剩下的值整体错位 + assert dict(rows[0]) == {c: recorded[c] for c in _PRE_135_WRITTEN_COLUMNS} + + assert [m for m in captured_warnings if "写入失败" in m] == [] + assert [m for m in captured_warnings if "补列失败" in m] == [] + assert recorder.telemetry_status.degraded is False + notices = [m for m in captured_warnings if "auto_migrate=False" in m] + assert len(notices) == 1 # 准备期一次讲清,不逐行刷屏 + # 逐字钉住十个维度: 前缀断言会让将来漏进告警的新列照样绿 + assert _PRE_135_MISSING_NOTICE in notices[0] + # 提示里的 SQL 必须可直接粘贴执行(首列与末列各验一条,含类型) + assert "ALTER TABLE llm_calls ADD COLUMN scope TEXT;" in notices[0] + assert "ALTER TABLE llm_calls ADD COLUMN total_latency_ms INTEGER;" in notices[0] + finally: + await recorder.aclose() + + async def test_old_and_new_writers_share_one_table(self, fresh_schema, captured_warnings): + """滚动升级期的混写: 已补列的表上,旧版本进程按 26 列写、新版本按 36 列写。 + + 这是升级窗口里必然出现的形态(先升一个 worker,其余仍是 1.3.4),而它的失败 + 方式是静默的: 若新列带了 NOT NULL 或旧列集的 INSERT 被拒,旧 worker 的遥测 + 会整段消失而只留逐行 warning。故这里既断三行都在、也断没有写入失败 warning。 + + 旧进程用 `insert_sql("postgres", 旧列集)` 而不是另抄一条 SQL: 1.3.4 的 + recorder 发出的正是同一函数按当年列集拼出的语句,另抄一份只会各自漂移。 + """ + fresh_dsn, schema = fresh_schema + recorder = _recorder(fresh_dsn, auto_migrate=True) + try: + # 新版本进程: 建表(37 列)并写一条带完整新列的终态行 + await _record_minimal( + recorder, + call_id="new-1", + scope="LLM", + operation="chat", + logical_call_id="lcid-mix", + event_kind="terminal_failure", + error_type="AllSourcesExhausted", + attempts=2, + total_latency_ms=1500, + ) + + # 旧版本进程: 同一张表,按 1.3.4 的 26 列集写入 + legacy_fields = _minimal_fields("old-1", response="from a 1.3.4 worker") + await _execute_args( + fresh_dsn, + insert_sql("postgres", _PRE_135_WRITTEN_COLUMNS), + *(legacy_fields[c] for c in _PRE_135_WRITTEN_COLUMNS), + ) + + # 新版本进程继续写: 旧进程的写入不得污染后续(列集是每进程各自探测的) + await _record_minimal( + recorder, call_id="new-2", scope="LLM", operation="embed", event_kind="attempt" + ) + + cols = await _fetch( + fresh_dsn, + "SELECT column_name FROM information_schema.columns " + "WHERE table_schema = $1 AND table_name = 'llm_calls' ORDER BY ordinal_position", + schema, + ) + assert [r["column_name"] for r in cols] == _EXPECTED_COLUMNS # 旧进程不改表结构 + + rows = await _fetch( + fresh_dsn, + "SELECT call_id, response, scope, operation, event_kind, attempts, " + "total_latency_ms FROM llm_calls ORDER BY call_id", + ) + assert [r["call_id"] for r in rows] == ["new-1", "new-2", "old-1"] + by_id = {r["call_id"]: r for r in rows} + assert by_id["old-1"]["response"] == "from a 1.3.4 worker" + # 旧进程写下的行,新列一律 NULL——它没写,也不该被谁替它填 + assert (by_id["old-1"]["scope"], by_id["old-1"]["event_kind"]) == (None, None) + assert (by_id["old-1"]["attempts"], by_id["old-1"]["total_latency_ms"]) == (None, None) + assert (by_id["new-1"]["attempts"], by_id["new-1"]["total_latency_ms"]) == (2, 1500) + assert by_id["new-2"]["operation"] == "embed" + assert [m for m in captured_warnings if "写入失败" in m] == [] + + # 下游可见变化的机械化依据: 新口径"计失败调用"按 event_kind 过滤, + # 混写期旧行(event_kind 为 NULL)既不会被误计成失败,也不会被误计成成功 + terminal = await _fetch( + fresh_dsn, + "SELECT count(*) AS n FROM llm_calls WHERE event_kind = 'terminal_failure'", + ) + assert terminal[0]["n"] == 1 + unclassified = await _fetch( + fresh_dsn, "SELECT count(*) AS n FROM llm_calls WHERE event_kind IS NULL" + ) + assert unclassified[0]["n"] == 1 + finally: + await recorder.aclose() + + _PHYSICAL_COLUMNS_SQL = ( "SELECT column_name FROM information_schema.columns " "WHERE table_schema = $1 AND table_name = 'llm_calls' ORDER BY ordinal_position" @@ -1000,7 +1337,7 @@ class TestPublishedSchemaScript: await _execute_script(fresh_dsn, script) actual = [r["column_name"] for r in await _fetch(fresh_dsn, _PHYSICAL_COLUMNS_SQL, schema)] - # 物理列 = 26 个 INSERT 字段 + 库从不显式写的 created_at;对着库常量比,不另抄一份 + # 物理列 = 36 个 INSERT 字段 + 库从不显式写的 created_at;对着库常量比,不另抄一份 assert set(actual) == set(COLUMNS) | {"created_at"} # 列序也不许漂: 新列必须排在 created_at 之后,否则新建库与 ALTER 升级的列序分叉 assert actual == _EXPECTED_COLUMNS From 067b15be48d0aaa301fa87fb17f8ab96e5c7e75d Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 12:17:50 -0400 Subject: [PATCH 08/10] docs: document logical call telemetry and migration impact Field counts come from inspect, not memory: record_llm_call takes 36 parameters, COLUMNS has 36 entries, the physical table has 37. - README: capability table says 36 fields and names the three row kinds; new section covers reading call_stats, the five SQL migration items, the attribution query and the storage-side upgrade - README/.env.example/ARCHITECTURE: error_body follows the summarize_body limit and the structured-exhaustion error carries its own bounded explanation, so neither is inside PGW_TELEMETRY_TEXT_CAP coverage - ARCHITECTURE 7.8: the ten columns with per-column semantics, the I3/I4 invariants, operation versus exc.operation, and the assembly gate - CHANGELOG: unreleased section listing the four public changes and what downstream must do, in particular counting failures by event_kind and the assembly-time error for custom recorders - schemas/llm-calls: the ten columns plus a three-row-kind section - metrics/call-telemetry-coverage: 1.3.5 coverage contract, real live baselines left unfilled rather than stating a fake percentage Validation record records the T4 evidence: mechanical migration red then green, the four PG acceptance cases, the seven-item mutation matrix with all seven killed and the copy restored to an identical digest, plus the PYTHONPATH pitfall that made the first mutation round silently test the original source. Version numbers and release steps are deliberately untouched. --- .env.example | 3 + CHANGELOG.md | 34 ++++++ README.md | 42 ++++++- research-wiki/ARCHITECTURE.md | 18 ++- ...09-09-135-call-observability-validation.md | 104 ++++++++++++++++-- .../metrics/call-telemetry-coverage.md | 14 +++ research-wiki/schemas/llm-calls.md | 35 +++++- 7 files changed, 237 insertions(+), 13 deletions(-) diff --git a/.env.example b/.env.example index e76b13c..590a271 100644 --- a/.env.example +++ b/.env.example @@ -113,6 +113,9 @@ PGW_TELEMETRY_BACKEND=none # sqlite | postgres | none(必填) # # 也要能拿原样的请求复现与重放;截断后这两件事都做不成,而既有下游正依赖这一行为。 # # 反面同样要看清: 不截断意味着客户合同、标书全文无限期留在 llm_calls 里, # # 多租户下还混在同一张表。真在意留存面的部署应显式设一个上限,并配保留期与访问控制。 +# # 1.3.5 补充: error_body(网关响应正文摘要)沿用库内 summarize_body 上限, +# # 结构化耗尽终态行的 error 说明自带有界限长(不含模型原始正文); +# # **两者都不在本键的覆盖面内**,估算留存面时要单独计。 # PGW_PRICING_PATH=config/prices.json # 可选: {"": {"input_per_1m": x, "output_per_1m": y}};缺省 cost 恒 None # # 可选第三档 "cached_input_per_1m": z —— 供应商 prompt cache 命中部分的单价; # # 不填即命中部分也按 input 全额计(库不猜折扣率),cost 会偏高 diff --git a/CHANGELOG.md b/CHANGELOG.md index 7d88ffe..7217668 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,39 @@ # Changelog +## 未发布 + +把治理单位从「一次尝试」补齐到「一次逻辑调用」(issue #19、#23)。此前重试、换源、结构化重问、embedding 分批都各自独立可见,而「这一次调用总共打了几次、总共花了多久、最后为什么失败」在库外拼不出来;结构化耗尽、embedding/OCR 的无源与准入拒绝更是**一条遥测行都没有**。 + +### 公共面四项变更 + +| # | 位置 | 变更 | 谁会当场断 | +| --- | --- | --- | --- | +| 1 | `polygateway.CallStats` | 新导出的 frozen dataclass(`logical_call_id` / `attempts` / `total_latency_ms`) | 无(纯新增) | +| 2 | `LLMResponse` / `EmbeddingResponse` / `OcrTextResult` / `OcrLayoutResult` | 各追加**末尾**字段 `call_stats: CallStats \| None = None` | 按位置解包这四个类型的代码 | +| 3 | `ports.TelemetryRecorder.record_llm_call()` | 新增 **10 个无默认值 keyword-only 参数**(26 → 36 参) | 任何自建 recorder——且**在装配期当场报错**,不再是运行期静默丢行 | +| 4 | 遥测表 `llm_calls` | 追加 10 列(INSERT 字段 26 → 36,物理列 27 → 37);新增 `event_kind='terminal_failure'` 行 | 按旧口径计失败调用数的 SQL | + +第 3 条的装配期报错是有意的:`_record` 的 `except Exception` 会把旧签名 recorder 的 `TypeError` 吞成 warning,后果是下游升级后 **100% 丢遥测且调用照常成功**。降级方向的铁律管的是运行期写失败,不是装配错误。`**fields` 形态的 recorder 不受影响。 + +### 下游必须做的事 + +| 动作 | 说明 | +| --- | --- | +| **计失败调用改 `WHERE event_kind = 'terminal_failure'`** | 一次逻辑调用恰好一条终态行。`error IS NOT NULL` 跨尝试行与终态行,升级后计数会变大 | +| `AVG(latency_ms)` 按 `event_kind` 分组 | 终态行的 `latency_ms` 是整个逻辑调用的总耗时,与单次尝试不同量纲 | +| 自建 recorder 补齐 10 参 | 或改成 `**fields`;同时同步自己的 schema 与 INSERT 字段 | +| PG manual 档下游补列 | `telemetry_schema_sql("postgres")` 自取;不补则这 10 个维度按现有列裁剪后静默不落库(库发一条点名警告) | +| 注意失败行可能带 `http_status_code = 200` | MonkeyOCR 的 `success != true` 就是 200 下的失败,该列不可当成败判据 | + +费用口径**不变**:终态行 `cost IS NULL`、`usage_source='unavailable'`、token 为 0,不参与 `SUM(cost)`。400 密集负载下错误行会翻倍(既有 attempt 错误行 + 新增终态行),这是已批准的下游可见变化。 + +### 其他 + +- 新列全部可空、无默认值、追加在现有末列之后;**旧行不回填**(NULL 表达「补列之前根本没记过这件事」),滚动升级期新旧进程可写同一张表。 +- `error_body` 沿用 `summarize_body` 上限,结构化耗尽终态行的 `error` 带有界说明(不含模型原始正文);**两者都不在 `PGW_TELEMETRY_TEXT_CAP` 覆盖面内**。 +- 修正历史误标:`embed()` 非 200 的失败现在报 `operation="embedding"`。遥测新列 `operation` 由调用点给定,与 `exc.operation` 是两个语义。 +- 重试预算与退避、429 免预算与 stall 算法、取消结算、限流/熔断语义、缓存 key 公式一字未改。 + ## 1.3.4(2026-09-09) > [!WARNING] diff --git a/README.md b/README.md index c007326..5515c75 100644 --- a/README.md +++ b/README.md @@ -20,7 +20,8 @@ | 流式看门狗 | TTFT / inter-token / 总超时三层活性;thinking token 刷活性不计结果;截断流(缺 `[DONE]`)判瞬时不入缓存 | | 推理可观测性 | "这次到底推理没推理"由多信号裁定(推理正文压倒 usage 明细),三态落在 `LLMResponse.thinking_observation`:`observed` / `absent` / `unknown`——**`unknown` 是"本次判不出",不是"没推理"**;本次实发档位与实测观测矛盾时按 `(源, 模型, 生效档位)` 各告警一次(能力表过期、开启未生效、注入了却观测不到;同一模型的 low 与 max 是两个独立的矛盾,不共用节流键);裁定结果随遥测落库 | | 推理档位 | 推理是**八档**(`none`/`auto`/`minimal`/`low`/`medium`/`high`/`xhigh`/`max`)而非开关:源级 `REASONING_EFFORT` + 请求级 `chat(reasoning_effort=...)`,`ENABLE_THINKING` 保留为语法糖;库带 24 条能力表(逐条 evidence 自报实测/文档推定),档位打空**默认报错并给出该模型最省的可用档与该配的键**,要静默映射需显式配 `EFFORT_FALLBACK=nearest`;实发档随 `LLMResponse.applied_effort` 与遥测落库 | -| 遥测与成本 | 每次调用(含缓存命中与失败)必录 26 字段;SQLite / Postgres 后端(表已存在时**不需要** schema 建表权限,最小权限账号可直接用);按价格表折算成本落库(注意 `LLMResponse.cost` 本身恒为 `None`,成本只进遥测);多模态内容摘要落库不存原图 | +| 遥测与成本 | 每次调用(含缓存命中与失败)必录 36 字段;三类行(`event_kind` = `attempt` / `cache_hit` / `terminal_failure`)加逐源诊断列(`http_status_code` / `error_type` / `cause_type` / `error_body`);SQLite / Postgres 后端(表已存在时**不需要** schema 建表权限,最小权限账号可直接用);按价格表折算成本落库(注意 `LLMResponse.cost` 本身恒为 `None`,成本只进遥测);多模态内容摘要落库不存原图 | +| 逻辑调用统计 | 治理单位是**一次逻辑调用**而非一次尝试:四种响应(chat / embedding / OCR 两种)带 `call_stats`(`logical_call_id` / `attempts` / `total_latency_ms`),重试、换源、结构化重问、embedding 分批共享同一逻辑 ID;每次**领域失败**另落一条 `terminal_failure` 行,失败调用数从此是一条 `WHERE event_kind = 'terminal_failure'`,详见[1.3.5 逻辑调用统计与失败诊断](#135-逻辑调用统计与失败诊断) | | 遥测的资源与降级 | Postgres 池**闲时占 0 条连接**、忙时上限可配(`PGW_TELEMETRY_PG_POOL_MAX`,缺省 4),每次写入有硬预算(`PGW_TELEMETRY_PG_WRITE_TIMEOUT_S`,缺省 5s);后端不可用是**可恢复的降级**(冷却 60s 后自动重试,DBA 建完表/放开权限即自愈),永久失能只留给 DSN 本身写错;降级状态可编程查询——`client.telemetry_status` 给出 `degraded`/`fatal`/`reason`/`dropped_rows` 等只读快照,不必再靠人工对账。**对账要同时看 `degraded` 与 `dropped_rows`**: 池饱和超预算丢的行走行级丢弃,`degraded` 保持 `False`(后端没挂,是本进程并发超了),只按 `degraded` 告警会看不见这一类丢行——而它恰是 `pool_max` 配小了的唯一信号 | | 调用方维度 | 每次调用可带 `tenant_id`(遥测表的真实列,可挂 RLS、可建复合索引)与 `meta`(≤16 个自定义 KV);四个公共方法全覆盖,校验超限即报错;**库只交付列,不启用 RLS、不建索引** | | 遥测表治理 | `llm_calls` 是**下游的表**:PG 侧缺省**不再自动 `ALTER` 补列**(`PGW_TELEMETRY_SCHEMA_MODE` 三态,不设则 sqlite→auto、postgres→manual),manual 档点名缺列并按现有列裁剪写入;`telemetry_schema_sql(backend)` 自取可粘进迁移文件的建表/补列 SQL;`PGW_TELEMETRY_TEXT_CAP` 限正文长度(**不设 = 存全文**);保留期与访问控制走[生产部署 DDL 模板](#生产部署-ddl-模板postgresql)加 `tools/telemetry_retention.py` | @@ -71,6 +72,43 @@ M8 包括 reasoning_effort、enable_thinking、thinking、thinking_budget、reas **本版验收例外(2026-09-09 用户正式批准)**:不再补全模型矩阵;既有失败、UNKNOWN、不可达、缺轮及下游现行配置缺证据如实保留,不改成 PASS。M2 两型的定向成功不代表全模型通过;三项目实际配置迁移仍未核验,合成兼容测试不能代替,缓存迁移操作前置也未被豁免。逐项实测、网络诊断与证据索引见[1.3.4 验证记录](research-wiki/findings/2026-09-09-134-thinking-contracts-validation.md)。 +## 1.3.5 逻辑调用统计与失败诊断 + +> [!WARNING] +> **升级后失败行会变多,旧的"失败调用数" SQL 会多数。** 每次领域失败除逐次尝试行外另写一条 `terminal_failure` 行;自建 recorder 若未跟进 10 个新参数,会在**装配期**就报错(不是运行期静默丢行)。 + +本版把治理单位从"一次尝试"补齐到"一次逻辑调用"。两件事: 四种响应带上 `call_stats`;遥测表补 10 列并新增一类行。 + +### 读 `call_stats` + +```python +resp = await client.chat([{"role": "user", "content": "hi"}]) +stats = resp.call_stats # EmbeddingResponse / OcrTextResult / OcrLayoutResult 同名字段 +stats.logical_call_id # 一次逻辑调用一个 ID:重试/换源/重问/分批共享 +stats.attempts # 真实打出去的尝试次数(免预算 429 也计;缓存命中为 0) +stats.total_latency_ms # 含缓存 IO、退避、准入等待、重问与内联遥测收尾的墙钟 +``` + +`call_stats` 为 `None` 意为**未知**(如第三方合成的响应、或旧缓存条目回放),不得读成 0。失败时异常对象上**不挂任何统计字段**——要归因请查遥测表。 + +### 下游 SQL 迁移五项 + +| # | 改什么 | 理由 | +| --- | --- | --- | +| 1 | 计失败调用改成 `WHERE event_kind = 'terminal_failure'` | 一次逻辑调用恰好一条终态行 | +| 2 | `error IS NOT NULL` 不再是失败调用的判据 | 它同时命中尝试错误行与终态行,升级后计数变大 | +| 3 | `AVG(latency_ms)` 须按 `event_kind` 分组 | 终态行的 `latency_ms` 是**整个逻辑调用**的总耗时,与单次尝试不同量纲 | +| 4 | 费用口径**不变** | 终态行 `cost IS NULL`、`usage_source='unavailable'`、token 为 0,不参与 `SUM(cost)` | +| 5 | 失败行可能带 `http_status_code = 200` | MonkeyOCR 的 `success != true` 就是 200 下的失败,该列不可当成败判据 | + +归因查询的典型形态: 一条 `WHERE logical_call_id = :lcid` 同时拿到整池终态原因(`terminal_failure` 行的 `error` / `error_type`)与逐源现场(`attempt` 行的 `source_name` / `http_status_code` / `cause_type` / `error_body`)。**终态行的 `http_status_code`/`cause_type`/`error_body` 恒为 NULL**: 拿最后一个源的现场冒充整池归因是错的。 + +### 存储侧升级 + +新增 10 列全部可空、无默认值、追加在现有末列之后(`scope`、`operation`、`logical_call_id`、`event_kind`、`http_status_code`、`error_type`、`cause_type`、`error_body`、`attempts`、`total_latency_ms`),INSERT 字段 26 → 36、物理列 27 → 37。auto 档自动追加;manual 档一条 DDL 不发、按现有列裁剪写入并点名缺列(补列 SQL 由 `telemetry_schema_sql("postgres")` 自取)。**旧行的新列一律 NULL 且不回填**——NULL 表达的是"补列之前根本没记过这件事";滚动升级期新旧进程可写同一张表。 + +自建 `TelemetryRecorder` 的下游请同步补齐这 10 个 keyword-only 无默认值参数(或改成 `**fields` 形态),并同步自己的 schema 与 INSERT 字段。装配期形状闸只能证明形状能被接受,证不了函数体真的落这些列。 + ## 安装 发布在实验室 Gitea PyPI(公开包,匿名可装): @@ -398,6 +436,8 @@ PGW_TELEMETRY_TEXT_CAP=2000 # 落库正文的字符上限;不设 = 存全 **`PGW_TELEMETRY_TEXT_CAP` 的覆盖面必须说清,否则合规判断会出错。** cap 落在四处:`messages` 里每条消息的字符串 `content`、多模态 content 数组中 `type == "text"` 的 part 的 `text`,以及 `response` 与 `thinking` 两列。消息侧的这个面与缓存摘要函数 `digest_messages` 一致——**只碰 `content`**,消息里别的字段一概不碰。所以调用方自己塞进 `tool_calls.function.arguments`、`name` 等字段的内容**不在覆盖范围内**:开了 cap 不等于表里没有全文残留。另需知道:缺省是**不截断**(存全文),而截断之后遥测不再是可复现重放的证据。 +**1.3.5 新增的两处诊断文本同样不在 cap 覆盖内**:`error_body` 记网关响应正文,沿用库内既有的 `summarize_body` 上限(与 cap 无关,也不随它变化);`error` 在结构化耗尽的终态行上带一段有界说明(修复原因至多 200 字符 + 至多 3 条校验错误 × 200 字符,**不含模型原始正文**),自带独立限长。判断留存面时这两列按各自上限估算,不能算进 cap 的四处。 + ### 7. SQLite 侧的保留期 SQLite 侧**不建议**对着一个大库文件跑 `DELETE` + `VACUUM`,而应**按天/按实验轮转库文件**——`runs/.db`、`runs/.db` 这样,到期直接删文件。这是三个现有下游(Video-Tree-TRM5 / CHSAnalyzer / dissect)天然就有的形态,比删行省事也安全得多:删文件是 O(1) 且不可能删错行,而 `VACUUM` 会重写整库、期间需要一倍磁盘空间,还会把并发写入方挡在外面。 diff --git a/research-wiki/ARCHITECTURE.md b/research-wiki/ARCHITECTURE.md index 5a733db..6eb6b3e 100644 --- a/research-wiki/ARCHITECTURE.md +++ b/research-wiki/ARCHITECTURE.md @@ -562,7 +562,7 @@ flowchart TB ### 7.8 遥测与成本 -**必录字段**(继承三项目 15 字段规范;当前 26 个 INSERT 字段,物理表列 27 = 26 + 数据库自填的 `created_at`,两套口径的区分见 `telemetry/schema.py` 模块 docstring): call_id、parent_call_id、session_id、model、provider、source_name、messages(JSON)、response、thinking、prompt_tokens、completion_tokens、usage_source、latency_ms、ttft_ms、max_inter_token_ms、cache_hit、error、**cost**、**cached_prompt_tokens**、**model_reported**、**sampling**、**reasoning_tokens**、**tenant_id**、**meta**、**thinking_observation**、**reasoning_effort**。 +**必录字段**(继承三项目 15 字段规范;当前 36 个 INSERT 字段,物理表列 37 = 36 + 数据库自填的 `created_at`,两套口径的区分见 `telemetry/schema.py` 模块 docstring): call_id、parent_call_id、session_id、model、provider、source_name、messages(JSON)、response、thinking、prompt_tokens、completion_tokens、usage_source、latency_ms、ttft_ms、max_inter_token_ms、cache_hit、error、**cost**、**cached_prompt_tokens**、**model_reported**、**sampling**、**reasoning_tokens**、**tenant_id**、**meta**、**thinking_observation**、**reasoning_effort**、**scope**、**operation**、**logical_call_id**、**event_kind**、**http_status_code**、**error_type**、**cause_type**、**error_body**、**attempts**、**total_latency_ms**。 **`sampling` 列(2026-07-31,issue #4,端口 20 → 21)**: 列语义 = 「调用方采样意图 ⊎ 生效源 `extra_body`」的 canonical JSON,空则 NULL。**不含**结构化注入的 `response_format`——列名是采样参数,schema 不是,且数 KB schema 逐行落库会让审计表无谓膨胀。三个 emit 入口口径必须各自定死,否则同一列在不同行含义不同: `emit_attempt`(RetryMW 调用,**唯一**有生效源者)并上 `source.extra_body`;`emit_cache_hit` / `emit_terminal_failure`(TelemetryMW 最外层调用)无 source 可言,只记调用级——与 `model`/`source_name` 在终态行置空是同一先例,且缓存命中行无损(`sampling` 已进缓存 key,能命中即意味调用级参数与历史那次逐字相同)。三者统一读 `request.sampling` 而非 `request.overlay`(后者在 RetryMW 处已被结构化注入污染、在 TelemetryMW 处未被污染,直接用必然三行分叉)。OCR/embedding 路径因决策 G 剥离 `extra_body`,该列恒 NULL。 @@ -581,6 +581,20 @@ flowchart TB 实际档分析须 `cache_hit=false AND error IS NULL`。embedding/OCR text/layout 的成功与失败尝试由 reasoning_applies=False 保证 NULL,真实 client→emitter→临时 SQLite 与 chat 阳性共同守卫,不能用空行集合证明。生产 emitter 单一出口、端口字段数和 DDL 不变。 +**逻辑调用十列(2026-09-09,issue #19/#23,端口 26 → 36)**: 本版把治理单位从"一次尝试"补齐到"一次逻辑调用"。十列按同一顺序追加在 `reasoning_effort` 之后,全部可空、无默认值、不回填旧行(NULL 表达的是"补列之前根本没记过这件事",与任何哨兵值不同;这一点与 `tenant_id` 故意相反——后者是 RLS 可见性需要哨兵,前者是归因需要区分真实缺口)。 + +| 列 | 语义 | +| --- | --- | +| `scope` | 池名,`TelemetryEmitter` 构造期注入(三个 client 各一行),三类行都带;**不拿 `source_name` 顶替**(终态失败可能根本没选出源) | +| `operation` | `chat` / `embed` / `recognize_text` / `parse_layout`,**由调用点给定**。它与 `PolyGatewayError.operation` 是两个语义: 后者是异常自报的出错环节(可为 `download_result` 这类子步骤),链路上任何位置不得读它来填本列 | +| `logical_call_id` | 一次逻辑调用一个 ID;上下文缺席(库内现场构造的请求)落 NULL,**不造 ID** | +| `event_kind` | 三态: `attempt` / `cache_hit` / `terminal_failure`,三类行的唯一机械判据 | +| `http_status_code` / `cause_type` / `error_body` | 只在失败的 `attempt` 行上非空(成功行不统一填 200: 那会让"有状态码"不再等价于"失败了");终态行三列恒 NULL(**C1 红线**) | +| `error_type` | 该行自身错误的类名;取消路径传字符串,故为 NULL | +| `attempts` / `total_latency_ms` | 只属终态行(同一份冻结快照,与该行 `latency_ms` 同值),其余两类行 NULL | + +两条不变量: **I3** —— 每次领域失败至多一条终态行(`_CallContext.claim_terminal()` 去重;recorder 写失败仍只 warning,故 SQL 可见 ≤ 1);**I4** —— 非领域异常(编程错)**零条**终态行、原样传播、分类不被改写。终态行仍 `cost=NULL`、`usage_source='unavailable'`、token 0,费用聚合口径不变。诊断值的提取收敛在 Emitter 内一个纯 helper(`_error_fields`),只读领域异常的既有属性,**不遍历任意对象、不解析字符串猜诊断**。另立**装配期形状闸**: 旧签名 recorder 在构造 `TelemetryEmitter` 时当场 `ValueError`——降级方向的铁律管的是**运行期写失败**,装配错误放行的后果是下游 100% 丢遥测且调用照常成功。 + **`thinking_observation` 列(2026-08-25,issue #16/#17,端口 24 → 25)**: 落 `LLMResponse.thinking_observation` 的裸取值(`observed` / `absent` / `unknown`,两端均为可空 `TEXT`),语义见 §5.1。它补的是 `reasoning_tokens` 补不上的那一格: 后者为 NULL 时「没推理」与「没上报」不可区分,而供应商停报 `completion_tokens_details` 是会真实发生的事(MiniMax 这一路 2026-08-25 实测已停报,qwen 与 deepseek 在同一网关同一 key 上照常返回),届时按 `reasoning_tokens IS NULL OR = 0` 统计「未推理」会把推理了的调用一并算进去。有了本列,口径改为按本列取值分组,`unknown` 独立成一档而不再被并进「未推理」。 **recorder 收到的必须是裸 `str` 而非枚举实例**: `TelemetryEmitter` 的 `_AttemptUsage` 内部持 `ThinkingObservation` 类型,`_record` 下沉时取 `.value`。`StrEnum` 虽是 `str` 子类,asyncpg 的参数编码对 `str` 子类不保证接受,而遥测写失败只降级为一条 warning——这类问题不会当场炸,只会让 Postgres 那一路悄悄少一列数据。归一化放在 emitter 侧,与 `tenant_id`/`meta`/`sampling` 由 emitter 定型后再交 recorder 是同一分工(recorder 只落库,不做语义判断)。列序纪律同上: 新列排在最末,两端 DDL 与两份 backfill 同步。 @@ -589,7 +603,7 @@ flowchart TB **schema 单一事实源、档位与冲突目标(2026-08-19,issue #13,决策见 D15)**: 列序、两端 DDL、两端补列语句、`INSERT` 构造与缺列告警收敛进 `telemetry/schema.py`——此前在两个 recorder 各存一份,而公共函数 `telemetry_schema_sql` 打印给下游的 SQL 必须与库真正执行的 DDL **同源**,三份必然漂移,漂移的表现是"下游照打印的 SQL 建完表,库仍报缺列"。补列自此由 `PGW_TELEMETRY_SCHEMA_MODE` 控制(三态: 不设按后端派生 sqlite→auto / postgres→manual,显式设置两侧均可覆盖): manual 档一条 DDL 都不发,改为按探测到的现有列**裁剪 `INSERT`**(裁剪是关掉 ALTER 的前提,否则缺列旧表每行写入都被拒 = 遥测全失)并发**一条**点名缺列、附可执行 SQL 的 warning;auto 档行为不变,且补列失败时**不裁剪**(该档承诺"把列补上",补不上就让缺列以逐行 warning 暴露)。**库内执行的补列语句与打印给人的那份是两套文本**: 库内不用 `ADD COLUMN IF NOT EXISTS`(它即便列已存在也先取 ACCESS EXCLUSIVE 锁,故库侧一律先探测后 ALTER),打印的那份带,以保证下游可重复执行。同批把 PG 写入的 `ON CONFLICT (call_id) DO NOTHING` 改为**无冲突目标**的 `ON CONFLICT DO NOTHING`: 带目标的语句要求恰好匹配 `(call_id)` 的唯一约束,而 PG 要求分区表的唯一约束必须包含分区键——按 `created_at` 分区(issue #12)后主键变成 `(call_id, created_at)`,该语句被 PG 直接拒收,而写失败只逐行 warning,表现为分区部署下遥测全线静默丢数据;无目标版本在两种表形态上都合法,普通表上语义逐字等价(表上只有主键这一个唯一约束),SQLite 的 `INSERT OR IGNORE` 本就无目标。 -**正文截断(2026-08-19,issue #12)**: `PGW_TELEMETRY_TEXT_CAP` 给落库正文一个可配置的字符上限,**缺省不设 = 不截断**(人类决策 E-a): 截断后的遥测不再是审计证据,也无法拿原样的请求复现与重放,而这正是既有下游在依赖的行为,默认改动即破坏;代价是 issue 那句"无限期保留全部租户全文不应是默认状态"只被解决一半——默认仍是全文,但下游第一次有了不写全文的手段。截断落在 `TelemetryEmitter._record`(全库唯一遥测出口,单一 helper 铁律)内,位于 `digest_messages` 之后、`json.dumps` 之前,作用面四处: 每条消息的字符串 `content`、多模态 part 中 `type == "text"` 的 `text`、`response`、`thinking`;超出部分头部硬切并附 `…(略 N 字)`。**按每条文本切而不是切整串 JSON**——后者会往不做任何校验的 TEXT 列里写进非法 JSON,让此后一切按 JSON 解析该列的分析全废。**且只产出新对象、绝不就地修改**: `digest_messages` 对非 list 的 `content` 原样透传同一个 dict 对象,就地截断会同时污染调用方持有的 messages、后续重试的请求体与缓存写入的 key 且全程无报错——红线由"cap 开与关两态下 `build_cache_key` 输出逐字节相同"的测试钉死。覆盖面须诚实声明: 只碰 `content`(与 `digest_messages` 处理面一致),调用方放进 `tool_calls.function.arguments` 等字段的内容不在其中。embedding 与 OCR 两条链路各自既有的 200 字符上限保留不动,与新 cap 是取更严者的关系。 +**正文截断(2026-08-19,issue #12)**: `PGW_TELEMETRY_TEXT_CAP` 给落库正文一个可配置的字符上限,**缺省不设 = 不截断**(人类决策 E-a): 截断后的遥测不再是审计证据,也无法拿原样的请求复现与重放,而这正是既有下游在依赖的行为,默认改动即破坏;代价是 issue 那句"无限期保留全部租户全文不应是默认状态"只被解决一半——默认仍是全文,但下游第一次有了不写全文的手段。截断落在 `TelemetryEmitter._record`(全库唯一遥测出口,单一 helper 铁律)内,位于 `digest_messages` 之后、`json.dumps` 之前,作用面四处: 每条消息的字符串 `content`、多模态 part 中 `type == "text"` 的 `text`、`response`、`thinking`;超出部分头部硬切并附 `…(略 N 字)`。**按每条文本切而不是切整串 JSON**——后者会往不做任何校验的 TEXT 列里写进非法 JSON,让此后一切按 JSON 解析该列的分析全废。**且只产出新对象、绝不就地修改**: `digest_messages` 对非 list 的 `content` 原样透传同一个 dict 对象,就地截断会同时污染调用方持有的 messages、后续重试的请求体与缓存写入的 key 且全程无报错——红线由"cap 开与关两态下 `build_cache_key` 输出逐字节相同"的测试钉死。覆盖面须诚实声明: 只碰 `content`(与 `digest_messages` 处理面一致),调用方放进 `tool_calls.function.arguments` 等字段的内容不在其中。embedding 与 OCR 两条链路各自既有的 200 字符上限保留不动,与新 cap 是取更严者的关系。**1.3.5 新增的两处诊断文本同样不在 cap 覆盖内**: `error_body` 沿用 `summarize_body` 的既有上限,结构化耗尽终态行的 `error` 说明自带有界限长(`repair=` 至多 200 字符 + `validation=` 至多 3 条 × 200 字符,**不含 `raw_text`**——模型正文的预算已由 attempt 行的 `response` 列承担)。两段限长常量与 `StructuredMW` 重问反馈共用同一份(`format_bounded_errors`),数值只有一份定义。 - 后端: `SQLiteRecorder`(默认;WAL + busy_timeout、`INSERT OR IGNORE` 幂等、`asyncio.to_thread` 桥接、初始化/写入失败全降级不冒泡)与 `PostgresRecorder`。 - **单一 helper 铁律**: 遥测调用点收敛为一个内部函数/上下文管理器;Video-Tree 与 GovDoc 各有 4-5 处逐字复制的 `record_llm_call(15 个参数)` 是本条的直接教训。 diff --git a/research-wiki/findings/2026-09-09-135-call-observability-validation.md b/research-wiki/findings/2026-09-09-135-call-observability-validation.md index a0a3270..5f371ad 100644 --- a/research-wiki/findings/2026-09-09-135-call-observability-validation.md +++ b/research-wiki/findings/2026-09-09-135-call-observability-validation.md @@ -1,7 +1,7 @@ --- type: finding node_id: finding:2026-09-09-135-call-observability-validation -title: "1.3.5 T2/T3 验收证据:36 列遥测、失败终态与取消口径" +title: "1.3.5 T2/T3/T4 验收证据:36 列遥测、失败终态、PG 存储兼容与变异矩阵" date: 2026-09-09 --- @@ -118,17 +118,94 @@ TDD 纪律要求"先失败后通过",且红必须是**行为红**而非 import 唯一顺带修正:`RetryMW.__init__` 的 `emitter` 注解由 `object | None` 收紧为 `TelemetryEmitter | None`(TYPE_CHECKING 导入,同层不破分层契约),因本轮改了它的 `_emit`。 -## 5. 明确缺口(交 T4 另任务,本轮不做) +## 5. T4 验收(PG 存储兼容、变异矩阵、文档同步) + +基线 HEAD `393f2bf`(1467 unit 全绿),全部命令在 `PolyGateway` conda 环境执行。 + +### 5.1 PG 存储兼容(真实实验室 Postgres,复用 `pg_sandbox`) + +机械迁移:`_EXPECTED_COLUMNS` 27 → 37 列、`_record_minimal` 字段字典补 10 键(与单测同款)、 +`_PRE_TENANT_COLUMNS` 的缺列集由4 列扩到 14 列。新十列一律由 `_CALL_OBSERVABILITY_COLUMNS` 派生, +两处 manual 档告警的逐字断言改成按 `COLUMNS` 序派生的 `_PRE_TENANT_MISSING_NOTICE`—— +另抄一份列名必然漂移,而漂移的表现是“manual 档没补列”这条断言假绿。 + +| 阶段 | 命令 | 结果 | +| --- | --- | --- | +| 机械迁移前(已知必红) | `pytest tests/integration/test_postgres_telemetry.py -q -x` | **1 failed / 18 passed**:manual 档列序断言 `Right contains 10 more items, first extra item: 'scope'` | +| 机械迁移后 | 同上(无 `-x`) | **27 passed**,EXIT=0,`tests/outputs/135/pg-telemetry-after-mechanical.log` | +| 新增四项验收后 | 同上 | **30 passed**,EXIT=0,`tests/outputs/135/pg-telemetry-new-cases.log` | + +新增 `TestCallObservabilityColumnsAcceptance`(对应计划 T4 的 PG 四项),共用一张 27 列的 1.3.4 形态旧表(`pre_135_schema`): + +| 验收项 | 用例与关键断言 | +| --- | --- | +| auto 追加 10 列 + 旧行 NULL | `test_pre_135_table_gains_the_ten_columns_and_old_rows_stay_null`:物理列 27 → 37 且列序 == `_EXPECTED_COLUMNS`;attempt 行与 terminal 行十列取值**整体比对**(不是逐条 in);历史行 `dict.fromkeys(...)` 十列全 NULL | +| manual 缺列裁剪 | `test_manual_trims_the_insert_on_a_pre_135_table`:表结构逐字不动(== `_PRE_135_COLUMNS`);裁剪后 26 列逐列等于提交值(防整体错位);恰 1 条告警且含可直接粘贴的首/末列 ALTER;无“写入失败”/“补列失败”,`degraded is False` | +| 新旧进程混写 | `test_old_and_new_writers_share_one_table`:新版建表写 36 列 → 旧版进程用 `insert_sql("postgres", 26 列集)` 写入 → 新版再写;三行共存、表结构不变、旧行新列全 NULL、无写入失败 warning | +| 下游口径验收 | 同上用例尾部:`WHERE event_kind = 'terminal_failure'` 计得 1;`event_kind IS NULL` 计得 1(混写期旧行既不误计成失败也不误计成成功) | + +纪律:未引用 `assert_no_leftovers`(它是 `test_pg_sandbox.py` 的模块级 fixture,对本文件不可见,上提它要改 `conftest.py`); +未新建沙箱设施、未碰共享表 `llm_calls`、未读或打印 DSN。新增的只有一个 `_execute_args`(带参数单语句)与 +`_minimal_fields`(从 `_record_minimal` 拆出的字段字典,供“旧进程”复用同一份取值)。 + +### 5.2 变异矩阵(仓库外副本,主工作区生产代码零改动) + +驱动脚本 `/tmp/pgw-135-mut/run_mutations.py`;每项“还原副本 → 施加单点变异(断言替换命中)→ 跑指定节点 → 还原”。 +完整日志 `tests/outputs/135/mutations.log`,脚本总退出码 **EXIT=0**。 + +| # | 变异 | exit | 被杀断言(节选) | +| --- | --- | --- | --- | +| M1 | `register_attempt()` 挪到 transport 成功之后 | 1 | `test_failed_retries_are_counted`、`test_budget_free_429_still_counts_as_an_attempt`、`test_retry_exhausted_counts_every_attempt` | +| M2 | `ChatRequest` 每次 `replace` 复制出新上下文 | 1 | `TestLogicalCallStats` 4 项 + `test_retry_exhaustion_writes_exactly_one_terminal_row` | +| M3 | `_rehydrate` 去掉 `call_stats=None` 覆盖 | 1 | `test_historic_dict_never_impersonates_call_stats` | +| M4 | Emitter 入口提前 `str(exc)` 压平 | 1 | 529/503 保真、4 个空超时文案 `cause_type`、类名前缀策略、结构化有界说明等 9 项 | +| M5 | 终态行复制最后一次 attempt 的 token/cost | 1 | `test_terminal_rows_never_contribute_to_cost`、`test_terminal_row_costs_nothing` | +| M6 | 去掉 `claim_terminal` 去重 | 1 | `test_terminal_row_is_written_once_per_logical_call`、`test_claim_terminal_is_true_once` | +| M7 | 装配闸改为捕获 `TypeError` 后 warning | 1 | `test_old_signature_recorder_is_refused_at_assembly`、`test_uninspectable_recorder_is_a_configuration_error` | + +还原校验:副本 `.py` 文件集合散列与纯净态**逐字一致**(`66b58b9a…`),还原后同一批节点 **172 passed**。 + +> **方法论陷阱(影响本仓所有变异证据的有效性)**:只设 `PYTHONPATH=<副本>/src` 是**无效的**—— +> `pyproject.toml` 的 `[tool.pytest.ini_options] pythonpath = ["src"]` 会把**仓库内**的 `src` 抢先塞进 `sys.path[0]`, +> 于是测的仍是原代码。首轮实跑七项变异**全部“存活”(exit=0)**就是这个坑; +> 改用 `-o pythonpath=<副本>` 覆盖后七项全部被杀。今后做变异必须先证“副本真的被导入”, +> 否则“变异存活”会被误读成“测试不够强”,而真相是变异根本没生效。 + +### 5.3 文档同步(字段数一律 `inspect` 实测,不凭记忆) + +实测值:`len(inspect.signature(TelemetryRecorder.record_llm_call).parameters) - 1 == 36`,`len(COLUMNS) == 36`,物理列 37。 + +| 位置 | 改了什么 | +| --- | --- | +| `README.md` 能力表 | “必录 26 字段” → 36 字段 + 三类行与诊断列;新增“逻辑调用统计”一行 | +| `README.md` 新小节 | 《1.3.5 逻辑调用统计与失败诊断》:`call_stats` 读法、**SQL 迁移五项**、归因查询、存储侧升级 | +| `README.md` cap 覆盖面 | 补“`error_body` 沿用 `summarize_body` 上限、结构化说明自带限长,**两者都不在 cap 覆盖内**” | +| `ARCHITECTURE.md` §7.8 | 必录字段 26 → 36(十列逐个列出);新增“逻辑调用十列”段(列语义表 + 不变量 I3/I4 + `operation` vs `exc.operation` 两个语义 + 装配闸);cap 段补两处诊断文本不在覆盖面 | +| `CHANGELOG.md` | 新建《未发布》:公共面四项变更表 + 下游必须做的事(点名 `WHERE event_kind` 与装配期报错) | +| `.env.example` | `PGW_TELEMETRY_TEXT_CAP` 块补 1.3.5 覆盖面例外 | +| `schemas/llm-calls.md` | 标题 26 → 36 字段;十列逐行登记;新增《三类行与失败归因口径》(含归因 SQL 与迁移四条) | +| `metrics/call-telemetry-coverage.md` | 新增《1.3.5 三类行与逻辑调用覆盖》;live 基线列**不写伪百分比** | + +版本号与发布步骤**未动**(任务边界);`pyproject.toml` / `__init__.py` 仍为 1.3.4。 + +### 5.4 本轮收口验证 + +| 检查 | 命令 | 结果 | +| --- | --- | --- | +| 静态 | `make check`(ruff format+lint + import-linter) | 见 §7 实跑记录 | +| 全量单测 | `pytest tests/unit -q` | 见 §7 | +| 目标集成 | `pytest tests/integration/test_postgres_telemetry.py -q` | **30 passed** | + +## 6. 剩余缺口(不归本任务,不自称已关) | 缺口 | 说明 | | --- | --- | -| PG 存储兼容 | `tests/integration/test_postgres_telemetry.py` 的 `_EXPECTED_COLUMNS` 与字段字典**尚未补 10 列**;auto 追加 / manual 缺列裁剪 / 旧行 NULL / 新旧进程混写四项未跑(需真实 PG 沙箱) | -| 变异证据 | 计划 §4 的七项变异(计数位置、上下文复制、缓存回放、提前压平、终态双计费用、去 `claim_terminal`、闸改 warning)未执行 | -| 文档同步 | README「必录 26 字段」、`PGW_TELEMETRY_TEXT_CAP` 覆盖面、ARCHITECTURE §7.8、CHANGELOG、schemas/metrics 未更新 | -| 独立验证 | 未派全新上下文 verifier(合并前硬门) | -| slow / e2e | 未跑,属发布清单第 4 步 | +| 独立验证 | 未派全新上下文 verifier(合并前硬门,由父会话前台派) | +| slow / e2e | 未跑,属发布清单第 4 步(CLAUDE §4.4.1) | +| live 覆盖基线 | `metrics/call-telemetry-coverage.md` 的实际基线列仍待首次生产运行填入 | +| 下游自建 recorder | 装配闸只能证明形状可被接受,证不了函数体真的落这些列(已写进 README/CHANGELOG) | -## 6. 环境噪音记录(不是缺陷) +## 7. 环境噪音记录(不是缺陷) 自动检查器用**系统解释器 Python 3.13.9**(无 `redis`/`pydantic`/`httpx`/`loguru`/`asyncpg` 等依赖), 持续报 `test_client.py` 2 项失败与大量 "Import could not be resolved"、`StrEnum is unknown import symbol`。 @@ -139,3 +216,14 @@ TDD 纪律要求"先失败后通过",且红必须是**行为红**而非 import - 项目强制环境 `conda run -n PolyGateway`(Python 3.12.13)下:`test_client.py` 98 passed、全量 1467 passed。 判据以 CLAUDE.md §2 规定的 conda 环境与 `make check` 为准。 + +**T4 轮次同样现象(已逐项复现并关闭)**:检查器报 `tests/integration/test_postgres_telemetry.py` “2/3 failed” +与 6 处 `Import "asyncpg" could not be resolved`。根因与上同:`asyncpg>=0.29` 是 optional extra +(`pyproject.toml:25` 的 `postgres`),检查器解释器里没装。已做确定性复现: + +- 那 3 条恰是本文件**仅有的不依赖真实 PG 的用例**,其中 2 条要造 `PostgresRecorder`; +- 用 `/home/iomgaa/miniconda3/bin/python`(无 asyncpg)跑这 3 条:**2 failed / 1 passed / 0.09s**,与检查器报告逐字吹合; +- 同 3 条在 `conda run -n PolyGateway` 下:**3 passed**;整文件 **30 passed**(真实 PG); +- 被标记的 6 行均为**本轮未触及的旧行**,本轮只新增 1 处同款函数内 `import asyncpg`。 + +未为此修改代码:给旧行加 type-ignore 属任务外改动(反 gold-plating),且并非真修复。 diff --git a/research-wiki/metrics/call-telemetry-coverage.md b/research-wiki/metrics/call-telemetry-coverage.md index a3f6077..9c89ce8 100644 --- a/research-wiki/metrics/call-telemetry-coverage.md +++ b/research-wiki/metrics/call-telemetry-coverage.md @@ -17,3 +17,17 @@ date: 2026-07-20 | live 覆盖 | 逐轮 PASS/FAIL/UNCOVERED、计划轮数与缺轮分别统计;必需单元不因 pytest exit 0 自动放行 | 未执行;UNKNOWN/缺轮/skip 不能记PASS | 复用 schema:llm-calls(无新字段/DDL),证据索引见 `findings/2026-09-09-134-thinking-contracts-validation.md`。独立错误取证只在 tests 内存,Markdown 只记录白名单安全摘要与布尔校验,不使用生产遥测旁路补失踪尝试。生产埋点仍是 TelemetryEmitter 单点出口。 + +## 1.3.5 三类行与逻辑调用覆盖 + +本版把覆盖度的计量单位从"一次尝试"改成"一次逻辑调用"——此前结构化耗尽、embedding/OCR 的无源与准入拒绝根本没有任何行,"每次调用必录"在这几条路径上是不成立的。 + +| 指标 | 确定性阈值/证据 | 实际 live 基线 | +| --- | --- | --- | +| 领域失败的终态行覆盖 | chat 结构化耗尽、embedding/OCR 的 `no_sources`、准入拒绝、`retry_exhausted`、尝试外取消各恰有 **1** 条;单测以真实临时 SQLite 抓实 | 待首次实际运行,不填伪百分比 | +| 至多一条(I3) | 同一逻辑调用重复进出口不产生第二条(`claim_terminal`);recorder 写失败仅 warning,故 SQL 可见 ≤ 1 | 待首次实际运行 | +| 非领域异常 0 条(I4) | 编程错(如 `KeyError`)不写终态行、原样传播、分类不被改写 | 待首次实际运行 | +| 诊断列保真 | 中转把 529 改写成 503 则记 503(**不猜回**);空 `str()` 的 httpx 超时类落 `cause_type`;成功行五列 NULL | 待首次实际运行 | +| 存储兼容 | 真实 PG 沙箱: auto 追加十列、manual 裁剪写入、旧行新列恒 NULL、新旧进程混写共存 | 2026-09-09 实跑 30 passed(见下方证据索引) | + +本版**新增十列**(见 schema:llm-calls),不新建表、不新增依赖。证据索引见 `findings/2026-09-09-135-call-observability-validation.md`;覆盖度真实基线仍待首次生产运行填入,不写伪百分比。 diff --git a/research-wiki/schemas/llm-calls.md b/research-wiki/schemas/llm-calls.md index 37bede5..1369b33 100644 --- a/research-wiki/schemas/llm-calls.md +++ b/research-wiki/schemas/llm-calls.md @@ -1,11 +1,11 @@ --- type: schema node_id: schema:llm-calls -title: "表结构: llm_calls(遥测 26 字段)" +title: "表结构: llm_calls(遥测 36 字段)" date: 2026-07-20 --- -# 表结构: llm_calls(遥测 26 字段) +# 表结构: llm_calls(遥测 36 字段) ## 列定义(冻结,M1 设计 §4.4 / ARCH §7.8) @@ -31,6 +31,37 @@ date: 2026-07-20 | meta | TEXT / JSONB NOT NULL DEFAULT '' / '{}' | 调用方自定义维度(同批,≤16 个 KV);SQLite 存 canonical JSON 串,PG 存 JSONB | | thinking_observation | TEXT | 本次推理是否真的发生的三态裁定(2026-08-25,issue #16/#17);`observed` / `absent` / `unknown`。见下方口径 | | reasoning_effort | TEXT | 按四种行来源记录的推理意图/实际编码档位(2026-09-09 澄清,issue #20/#26);八档 `Effort` 字面量之一,NULL = 调用方未表态(与 `none`「明确要求不推理」不可混同)。见下方口径 | +| scope | TEXT | 池名(2026-09-09,issue #19/#23);Emitter 构造期注入,三类行都带,**不拿 `source_name` 顶替** | +| operation | TEXT | `chat` / `embed` / `recognize_text` / `parse_layout`,由调用点给定;与 `PolyGatewayError.operation` 是两个语义,链路上不得读后者填本列 | +| logical_call_id | TEXT | 一次逻辑调用一个 ID(重试/换源/重问/分批共享);上下文缺席落 NULL,**不造 ID** | +| event_kind | TEXT | 三态 `attempt` / `cache_hit` / `terminal_failure`——三类行的唯一机械判据。见下方口径 | +| http_status_code | INTEGER | 失败 attempt 行的实收状态码(中转改写过就记改写后的,**不猜回原值**);成功行与终态行恒 NULL,且 200 也可能是失败行(MonkeyOCR `success != true`) | +| error_type | TEXT | 该行自身错误的类名;取消路径传字符串故为 NULL | +| cause_type | TEXT | `__cause__` 的类名(如 `ReadTimeout`)——httpx 超时类的 `str()` 为空,只靠 error 列分不出是哪种超时;仅失败 attempt 行非空 | +| error_body | TEXT | 网关响应正文摘要(`summarize_body` 上限,**不在 `PGW_TELEMETRY_TEXT_CAP` 覆盖面内**);仅失败 attempt 行非空 | +| attempts | INTEGER | 该逻辑调用真实打出去的尝试次数(免预算 429 也计);**只属终态行** | +| total_latency_ms | INTEGER | 该逻辑调用的总墙钟(含缓存 IO、退避、准入等待、重问);**只属终态行**,与该行 `latency_ms` 同取一份冻结快照 | + +## 三类行与失败归因口径(2026-09-09,issue #19/#23) + +遥测行不再只有"一次尝试"一种。`event_kind` 三态同时决定其余列的取值面: + +| event_kind | 什么时候写 | 列取值 | +| --- | --- | --- | +| `attempt` | 每次真实尝试(含失败与取消) | 失败时诊断四列可非空;`attempts`/`total_latency_ms` NULL | +| `cache_hit` | 缓存命中 | 诊断四列与快照两列全 NULL | +| `terminal_failure` | 每次**领域失败**的整池终态,每逻辑调用至多一条 | `attempts`/`total_latency_ms` 非空;`http_status_code`/`cause_type`/`error_body` 恒 NULL;`cost` NULL、`usage_source='unavailable'`、token 0 | + +两条不变量: 每次领域失败至多一条终态行(I3,`claim_terminal()` 去重);非领域异常(编程错)**零条**终态行、原样传播(I4)。 + +**终态行三列恒 NULL 是红线**: 把最后一次 attempt 的状态码与正文搬上来,等于拿最后一个源冒充整池归因。逐源现场由同一 `logical_call_id` 的 attempt 行给出: + +```sql +SELECT event_kind, source_name, http_status_code, error_type, cause_type, error, error_body +FROM llm_calls WHERE logical_call_id = :lcid ORDER BY created_at; +``` + +下游口径迁移四条: ① 计失败调用改 `WHERE event_kind = 'terminal_failure'`;② `error IS NOT NULL` 不再是失败调用判据(跨两类行);③ `AVG(latency_ms)` 须按 `event_kind` 分组(终态行是整个逻辑调用的总耗时);④ 费用口径不变(终态行 cost 恒 NULL)。 ## usage/成本口径(2026-07-30,est_tokens 解耦) From b4812e12c8dae174547017d6b50499e94e629d06 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 12:50:51 -0400 Subject: [PATCH 09/10] fix: degrade terminal telemetry failures instead of masking domain errors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 终态出口 `emit_terminal_once` 此前只让 `_record` 内的 except 兜住落库,而 诊断字段的提取(`_error_fields` → `_structured_detail` → `format_bounded_errors`) 在降级 try **之外**求值。下游经公共端口(自实现 `StructuredOutputStrategy` 或 transport)构造出 `ResultInvalidError(validation_errors=(非 str,))` 时,提取期 抛的 `TypeError` 会顶替调用方本该收到的领域异常——错误四分类被击穿(下游 `except ResultInvalidError` 落空),且 `claim_terminal()` 已消耗故终态行照样丢, 同时违反"遥测写失败降级不冒泡"。 改法与 RetryMW 的 attempt 出口(`retry.py::_emit`)同款: 把快照冻结与 await 整段包进 try,`CancelledError` 原样上抛、其余落一条 warning。终态行按已批准的 best effort(兜底命中时该次逻辑调用 0 条终态行,不补写)。异常类型校验与 `ResultInvalidError` 的既有设计均未改动。 顺带同步审查报告的 Minor 项: SQLite recorder docstring 26 → 36 字段、 research-wiki 索引重建、ARCHITECTURE 的 `sampling` 段落终态调用点口径, 并删除 `TelemetryMW` 迁移后无读取点的 `self._now` 死字段(保留形参, 避免平白打断既有装配写法)。 --- research-wiki/ARCHITECTURE.md | 2 +- research-wiki/index.md | 7 +-- research-wiki/log.md | 1 + src/polygateway/middleware/telemetry.py | 37 +++++++++---- src/polygateway/telemetry/sqlite.py | 2 +- tests/unit/test_client.py | 72 +++++++++++++++++++++++++ 6 files changed, 105 insertions(+), 16 deletions(-) diff --git a/research-wiki/ARCHITECTURE.md b/research-wiki/ARCHITECTURE.md index 6eb6b3e..88512d0 100644 --- a/research-wiki/ARCHITECTURE.md +++ b/research-wiki/ARCHITECTURE.md @@ -564,7 +564,7 @@ flowchart TB **必录字段**(继承三项目 15 字段规范;当前 36 个 INSERT 字段,物理表列 37 = 36 + 数据库自填的 `created_at`,两套口径的区分见 `telemetry/schema.py` 模块 docstring): call_id、parent_call_id、session_id、model、provider、source_name、messages(JSON)、response、thinking、prompt_tokens、completion_tokens、usage_source、latency_ms、ttft_ms、max_inter_token_ms、cache_hit、error、**cost**、**cached_prompt_tokens**、**model_reported**、**sampling**、**reasoning_tokens**、**tenant_id**、**meta**、**thinking_observation**、**reasoning_effort**、**scope**、**operation**、**logical_call_id**、**event_kind**、**http_status_code**、**error_type**、**cause_type**、**error_body**、**attempts**、**total_latency_ms**。 -**`sampling` 列(2026-07-31,issue #4,端口 20 → 21)**: 列语义 = 「调用方采样意图 ⊎ 生效源 `extra_body`」的 canonical JSON,空则 NULL。**不含**结构化注入的 `response_format`——列名是采样参数,schema 不是,且数 KB schema 逐行落库会让审计表无谓膨胀。三个 emit 入口口径必须各自定死,否则同一列在不同行含义不同: `emit_attempt`(RetryMW 调用,**唯一**有生效源者)并上 `source.extra_body`;`emit_cache_hit` / `emit_terminal_failure`(TelemetryMW 最外层调用)无 source 可言,只记调用级——与 `model`/`source_name` 在终态行置空是同一先例,且缓存命中行无损(`sampling` 已进缓存 key,能命中即意味调用级参数与历史那次逐字相同)。三者统一读 `request.sampling` 而非 `request.overlay`(后者在 RetryMW 处已被结构化注入污染、在 TelemetryMW 处未被污染,直接用必然三行分叉)。OCR/embedding 路径因决策 G 剥离 `extra_body`,该列恒 NULL。 +**`sampling` 列(2026-07-31,issue #4,端口 20 → 21)**: 列语义 = 「调用方采样意图 ⊎ 生效源 `extra_body`」的 canonical JSON,空则 NULL。**不含**结构化注入的 `response_format`——列名是采样参数,schema 不是,且数 KB schema 逐行落库会让审计表无谓膨胀。三个 emit 入口口径必须各自定死,否则同一列在不同行含义不同: `emit_attempt`(RetryMW 调用,**唯一**有生效源者)并上 `source.extra_body`;`emit_cache_hit` / `emit_terminal_failure`(前者由 TelemetryMW 最外层调用,后者自 1.3.5 起由三个 client 的公开边界经 `emit_terminal_once` 统一写出)无 source 可言,只记调用级——与 `model`/`source_name` 在终态行置空是同一先例,且缓存命中行无损(`sampling` 已进缓存 key,能命中即意味调用级参数与历史那次逐字相同)。三者统一读 `request.sampling` 而非 `request.overlay`(后者在 RetryMW 处已被结构化注入污染、在 TelemetryMW 处未被污染,直接用必然三行分叉)。OCR/embedding 路径因决策 G 剥离 `extra_body`,该列恒 NULL。 **`reasoning_tokens` 列(2026-08-11,issue #6,端口 21 → 22)**: 推理 token 已计入 `completion_tokens`,故成本总额一直是对的——这不是计费缺口而是**归因**缺口:缺了它,"这次调用花的钱里有多少花在推理上"无法区分,也就无从判断某个 scope 该不该关推理。供应商不报时记 NULL 而非 0(不可得 ≠ 为零,与 `usage_source='unavailable'` 同一纪律)。 diff --git a/research-wiki/index.md b/research-wiki/index.md index 471d247..7bf1519 100644 --- a/research-wiki/index.md +++ b/research-wiki/index.md @@ -1,6 +1,6 @@ # Research Wiki 索引 -> 自动生成,更新时间:2026-09-09 13:42 UTC +> 自动生成,更新时间:2026-09-09 16:49 UTC ## design (43) @@ -48,9 +48,10 @@ - [调用方自定义维度设计(issue #11)](designs/issue11-caller-dimensions.md) `design:issue11-caller-dimensions` - [采样参数透传设计(issue #4)](designs/sampling-params.md) `design:sampling-params` -## finding (15) +## finding (16) - [1.3.4 推理契约验证与发布准备](findings/2026-09-09-134-thinking-contracts-validation.md) `finding:2026-09-09-134-thinking-contracts-validation` +- [1.3.5 T2/T3/T4 验收证据:36 列遥测、失败终态、PG 存储兼容与变异矩阵](findings/2026-09-09-135-call-observability-validation.md) `finding:2026-09-09-135-call-observability-validation` - [2026-07-20-m2-soak-workload](findings/2026-07-20-m2-soak-workload.md) `finding:2026-07-20-m2-soak-workload` - [2026-07-21-m25-acceptance](findings/2026-07-21-m25-acceptance.md) `finding:2026-07-21-m25-acceptance` - [2026-07-21-p6-soak-baseline](findings/2026-07-21-p6-soak-baseline.md) `finding:2026-07-21-p6-soak-baseline` @@ -113,7 +114,7 @@ ## schema (1) -- [表结构: llm_calls(遥测 26 字段)](schemas/llm-calls.md) `schema:llm-calls` +- [表结构: llm_calls(遥测 36 字段)](schemas/llm-calls.md) `schema:llm-calls` ## metric (2) diff --git a/research-wiki/log.md b/research-wiki/log.md index d0fa0ca..6abfff4 100644 --- a/research-wiki/log.md +++ b/research-wiki/log.md @@ -158,3 +158,4 @@ - [2026-09-09 06:32 UTC] 重建索引: 98 篇页面 - [2026-09-09 13:42 UTC] 新增边: plan:2026-09-09-135-call-observability --implements--> design:2026-09-09-135-call-observability-design - [2026-09-09 13:42 UTC] 重建索引: 100 篇页面 +- [2026-09-09 16:49 UTC] 重建索引: 101 篇页面 diff --git a/src/polygateway/middleware/telemetry.py b/src/polygateway/middleware/telemetry.py index d728313..ef07fb4 100644 --- a/src/polygateway/middleware/telemetry.py +++ b/src/polygateway/middleware/telemetry.py @@ -679,7 +679,16 @@ async def emit_terminal_once( """三个 client 共用的**终态唯一出口**: 去重 + 同步冻结快照 + best effort 写入。 去重由 `claim_terminal()` 承担(每逻辑调用至多一条终态行);`emitter is None` - 或已写过 → 直接返回。写入侧异常按既有降级只落 warning(在 `_record` 内)。 + 或已写过 → 直接返回。 + + **降级范围包含诊断字段的提取与构建**,不只是写入那一步: `_record` 内的 + `except Exception` 只兜住落库,而 `_error_fields` / `canonical_sampling_json` + 在它**之外**求值——下游经公共端口(自实现 `StructuredOutputStrategy` 或 + transport)构造出的 `ResultInvalidError(validation_errors=(非 str,))` 会让提取期 + 抛 `TypeError` 顶替调用方本该收到的领域异常,错误四分类被击穿且终态行照样丢。 + 故在此整段兜底,与 RetryMW 的 attempt 出口(`retry.py::_emit`)同款写法。 + 终态行按 best effort: 兜底命中时本次逻辑调用**没有**终态行(`claim_terminal()` + 已消耗,不补写、不重试——重写一遍只会把同一个提取期异常再抛一次)。 **`CancelledError` 原样传播**(取消优先,不 shield、不开后台任务): 这一次 `await` 本身就是新的取消点,外部取消落在它上时调用方会看到 `CancelledError` @@ -688,15 +697,20 @@ async def emit_terminal_once( """ if emitter is None or not context.claim_terminal(): return - stats = context.snapshot() - await emitter.emit_terminal_failure( - request=request, - call_id=str(uuid.uuid4()), - error=error, - operation=operation, - stats=stats, - class_prefixed_error=class_prefixed_error, - ) + try: + stats = context.snapshot() + await emitter.emit_terminal_failure( + request=request, + call_id=str(uuid.uuid4()), + error=error, + operation=operation, + stats=stats, + class_prefixed_error=class_prefixed_error, + ) + except asyncio.CancelledError: + raise + except Exception as exc: + logger.warning("终态遥测记录失败(降级不冒泡): {}", exc) class TelemetryMW: @@ -711,7 +725,8 @@ class TelemetryMW: self, emitter: TelemetryEmitter, now: Callable[[], float] = time.monotonic ) -> None: self._emitter = emitter - self._now = now + # `now` 自 1.3.5 起本类不再读取(终态行迁到公开边界后无耗时可测),但形参保留: + # 删它会平白打断 `TelemetryMW(emitter, now=...)` 这一既有装配写法 async def __call__(self, request: ChatRequest, call_next: CallNext) -> LLMResponse: response = await call_next(request) diff --git a/src/polygateway/telemetry/sqlite.py b/src/polygateway/telemetry/sqlite.py index 7781530..c3e61dd 100644 --- a/src/polygateway/telemetry/sqlite.py +++ b/src/polygateway/telemetry/sqlite.py @@ -143,7 +143,7 @@ class SQLiteRecorder: logger.warning("SQLite 遥测补列失败(写入将逐行降级): {}", exc) async def record_llm_call(self, **fields: object) -> None: - """写一行遥测;字段集合即 26 字段冻结签名(ports.TelemetryRecorder)。 + """写一行遥测;字段集合即 36 字段冻结签名(ports.TelemetryRecorder)。 取值按 `self._columns`(manual 档可能已被裁剪),与 `self._insert` 的 占位符同序——两者必须一起改,分开改就是把值写进错位的列。 diff --git a/tests/unit/test_client.py b/tests/unit/test_client.py index 4213097..c9451ce 100644 --- a/tests/unit/test_client.py +++ b/tests/unit/test_client.py @@ -1645,3 +1645,75 @@ class TestChatTerminalFailureRows: operation="chat", ) assert len(self._rows(recorder, "terminal_failure")) == 1 + + +@pytest.fixture +def captured_warnings(): + """捕获库发出的 WARNING;loguru 不经标准 logging,pytest 的 caplog 抓不到。 + + 名字避开裸 `warnings`: 那会遮蔽标准库模块名,本文件将来任何一次 + `import warnings` 都会与它静默互相顶掉,而报错点离真因很远。 + """ + from loguru import logger + + messages: list[str] = [] + sink_id = logger.add(messages.append, level="WARNING") + yield messages + logger.remove(sink_id) + + +class TestTerminalEmitDegradation: + """终态出口的降级方向: 诊断字段的**提取**同样在降级范围内(铁律"遥测写失败降级不冒泡")。 + + 改前 `_record` 的 except 只包住写入本身,而错误诊断列在它之外求值—— + 经公共扩展点(自实现 `StructuredOutputStrategy`/transport)构造出的 + `ResultInvalidError(validation_errors=(非 str,))` 会让提取期抛 `TypeError` + 顶替调用方本该收到的领域异常,错误四分类当场被击穿。 + """ + + _MSG = [{"role": "user", "content": "hi"}] + + def _rows(self, recorder, kind): + return [r for r in recorder.rows if r["event_kind"] == kind] + + async def test_broken_validation_errors_keep_the_domain_error(self, captured_warnings): + """自实现策略给出非 str 的 `validation_errors`: 领域异常必须原样上抛。""" + + class _BadStrategy: + """公共端口 `StructuredOutputStrategy` 的下游实现(库外没有类型执法)。""" + + def request_overlay(self, schema): + return {} + + def parse(self, text): + raise ResultInvalidError("模型返回不可解析", validation_errors=(object(),)) + + recorder = _MemoryRecorder() + async with _client(telemetry=recorder, structured_strategy=_BadStrategy()) as client: + with pytest.raises(ResultInvalidError): # 不是 TypeError + await client.chat(self._MSG, structured="json") + # 降级有声: 静默吞掉等于遥测缺口无人知道 + assert [m for m in captured_warnings if "终态遥测记录失败" in m] + # 尝试行不受影响;终态行按 best effort 允许 0 条,但绝不能重复 + assert len(self._rows(recorder, "attempt")) == 1 + assert len(self._rows(recorder, "terminal_failure")) <= 1 + + async def test_cancellation_is_never_swallowed_by_the_degradation(self): + """降级不得吞取消: 写入那一次 await 上被取消,`CancelledError` 照常传播。""" + from polygateway.middleware.telemetry import emit_terminal_once + from polygateway.types import _CallContext + + class _CancellingEmitter: + async def emit_terminal_failure(self, **kwargs): + raise asyncio.CancelledError + + context = _CallContext(now=asyncio.get_running_loop().time) + request = ChatRequest(messages=self._MSG, call_context=context) + with pytest.raises(asyncio.CancelledError): + await emit_terminal_once( + _CancellingEmitter(), + request=request, + context=context, + error=AllSourcesExhausted(scope="llm", reason="retry_exhausted", retry_after_s=1.0), + operation="chat", + ) From 433039be795ff2dae8e15caff85db498a503a997 Mon Sep 17 00:00:00 2001 From: iomgaa Date: Wed, 9 Sep 2026 13:21:40 -0400 Subject: [PATCH 10/10] chore: prepare release 1.3.5 Version bump in pyproject and __init__, CHANGELOG dated 2026-09-09, README install lower bound raised to >=1.3.5, and release-prep evidence (remote check, gates, real gateway smoke and one bounded live probe) recorded in the 1.3.5 validation finding. --- CHANGELOG.md | 2 +- README.md | 2 +- pyproject.toml | 2 +- ...09-09-135-call-observability-validation.md | 61 +++++++++++++++++++ src/polygateway/__init__.py | 2 +- 5 files changed, 65 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7217668..5a9658b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,6 +1,6 @@ # Changelog -## 未发布 +## 1.3.5(2026-09-09) 把治理单位从「一次尝试」补齐到「一次逻辑调用」(issue #19、#23)。此前重试、换源、结构化重问、embedding 分批都各自独立可见,而「这一次调用总共打了几次、总共花了多久、最后为什么失败」在库外拼不出来;结构化耗尽、embedding/OCR 的无源与准入拒绝更是**一条遥测行都没有**。 diff --git a/README.md b/README.md index 5515c75..a1f3f3c 100644 --- a/README.md +++ b/README.md @@ -115,7 +115,7 @@ stats.total_latency_ms # 含缓存 IO、退避、准入等待、重 ```bash pip install --extra-index-url https://gitea.iomgaa.online/api/packages/iomgaa/pypi/simple/ \ - "polygateway[redis,postgres,structured]>=1.3.4,<2" + "polygateway[redis,postgres,structured]>=1.3.5,<2" ``` 核心仅依赖 `httpx` + `pydantic`;按需选 extras: diff --git a/pyproject.toml b/pyproject.toml index b36e9f4..35938a9 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "polygateway" -version = "1.3.4" +version = "1.3.5" description = "PolyGateway:实验室统一的大语言模型(LLM/VLM/OCR)调度与中转库——多源、限流、重试、熔断、缓存、遥测" # registry 包页面的正文只认这一项:缺了页面就是一片空白(1.1.2 的教训,twine 会警告 # long_description missing 但不阻塞上传)。README 在打包时被固化进产物,发布后再改无效。 diff --git a/research-wiki/findings/2026-09-09-135-call-observability-validation.md b/research-wiki/findings/2026-09-09-135-call-observability-validation.md index 5f371ad..5d65e10 100644 --- a/research-wiki/findings/2026-09-09-135-call-observability-validation.md +++ b/research-wiki/findings/2026-09-09-135-call-observability-validation.md @@ -227,3 +227,64 @@ TDD 纪律要求"先失败后通过",且红必须是**行为红**而非 import - 被标记的 6 行均为**本轮未触及的旧行**,本轮只新增 1 处同款函数内 `import asyncpg`。 未为此修改代码:给旧行加 type-ignore 属任务外改动(反 gold-plating),且并非真修复。 + +## 8. 1.3.5 发布准备(2026-09-09,唯一 writer 会话) + +范围仅**发布准备文档与版本号**:README 安装下界、CHANGELOG 定版、两处版本、本节记录。生产代码与测试**零改动**(`git diff` 只有 `CHANGELOG.md` / `README.md` / `pyproject.toml` / `__init__.py` 各 1 行,外加本文件追加的这一节)。未 merge/push/tag/构建/上传。 + +### 8.1 先查远端占用(改文件之前) + +| 检查 | 结果 | +| --- | --- | +| `git ls-remote --tags origin` | 最高 `refs/tags/v1.3.4`,**无 v1.3.5**;远端分支只有 `main`(`af57f93`)与 `docs/1.3.4-release-evidence` | +| 本地 tag | 同样止于 `v1.3.4` | + +远端未被占用是**本次时点**的事实,父会话真正 push/tag 前仍须复查以防竞态。 + +### 8.2 版本与文档数字(数字一律实测) + +| 项 | 结果 | +| --- | --- | +| 两处版本 | `pyproject.toml` 与 `src/polygateway/__init__.py` 同为 `1.3.5`;由 `tests/unit/test_package.py` 机械断言一致(**6 passed**) | +| README 安装下界 | `>=1.3.4,<2` → `>=1.3.5,<2`(本版含装配期形状闸与 SQL 口径迁移,旧下界会让下游装到不含新列的包) | +| README 迁移醒目度 | 《1.3.5 逻辑调用统计与失败诊断》为顶层小节且带 `[!WARNING]`(点名"失败行变多、旧失败 SQL 会多数、自建 recorder 装配期报错"),能力表"逻辑调用统计"一行直链该锚点;沿用 1.3.4 的同款版式,未重排既有 1.3.4 迁移节 | +| CHANGELOG 定版 | `## 未发布` → `## 1.3.5(2026-09-09)`,日期取自本机 `date +%F` 实际值,正文未改 | +| 字段数复核 | `inspect.signature(TelemetryRecorder.record_llm_call)` 去 self **36 参**、`len(schema.COLUMNS) == 36`、探针实测物理列 **37**——README/CHANGELOG 的 36/37 与实测吻合,非凭记忆 | + +README 静态数字未发现错误,故未做任何顺带修改。 + +### 8.3 复用既有证据的适用边界(不拿旧证据顶替本版统计) + +| 证据 | 是否复用 | 边界 | +| --- | --- | --- | +| 独立验证(全新上下文 verifier,父会话前台派) | 复用结论:**0 阻塞** | 审的是本分支 `b4812e1` 的 1.3.5 变更面;本会话未重复派子代理,也未把它当作 live 覆盖或发布后检查的替代 | +| 本版最终全套件 `make test` | **本版实跑,不复用** | `final-gates/make-test.log/.exit`:**1605 passed / 23 skipped / 108 deselected / 95%,exit 0** | +| 本版真实 PG 集成 | **本版实跑,不复用** | `final-gates/pg-telemetry-verbose.log/.exit`:**30 passed,exit 0**(含新增四项列兼容验收) | +| 本版 unit | **本版实跑** | 本轮版本号改动后复跑 `pytest tests/unit -q`:**1469 passed,exit 0**(`release/unit.log/.exit`) | +| 1.3.4 模型矩阵/逐型号推理证据 | **按适用条件复用** | 1.3.5 对 `thinking.py`、能力表、wire 片段、缓存 key 公式、重试/限流/熔断语义**一字未改**(§4 已逐文件确认),故 1.3.4 的型号级结论在其原有边界内继续成立——连同它的 FAIL/UNKNOWN/不可达/缺轮**一并继承**,不因本版而升格。它**不能**充当 1.3.5 新增 API/schema 的证据,也不提供本版的测试统计数字 | +| 1.3.4 的全模型矩阵重跑 | **不跑** | 本版未改推理路径;用户已批准不再补全模型矩阵。本会话不发全型号请求 | + +本版**改变的公共面**(`CallStats` 与四类响应新字段、端口 10 新参、遥测 10 列与 `terminal_failure` 行)全部由新写的离线真链路用例(三个 client + MockTransport 直到落库)、真实 SQLite 断言、真实 PG 30 项承担,见 §2、§2.4、§5.1。 + +### 8.4 本轮受影响路径的真实现场(并发 1、超时不压、无全模型请求) + +| 亲跑 | 命令与证据 | 结果 | +| --- | --- | --- | +| 真实网关冒烟(复用既有 e2e,不另造工具) | `pytest tests/e2e/test_smoke_gateway.py -m slow -v`,tmux `pgw135-smoke`,`release/smoke-gateway.log/.exit` | **4 passed,exit 0**(流式/非流式/结构化 json/结构化模型四节点,9.55s) | +| 有界现场探针(**恰 1 次**真实调用,全程走库) | `release/probe_call_stats.py` → `release/probe-call-stats.log/.exit`,tmux `pgw135-probe` | **exit 0**:`MiniMax-M3`/源 `minimax_1`,`call_stats.attempts=1`、`total_latency_ms=2581` ≥ 单次尝试 `2562`、`logical_call_id` 与落库行逐字一致;临时 SQLite 物理列 **37**、写入面 **36**、恰 1 条 `event_kind='attempt'` 行、`operation='chat'`、`scope='LLM'`、逻辑两列与四个诊断列全 NULL | + +探针脚本只落在 `tests/outputs/135/release/`(该目录 gitignore,不入库),不新增生产代码、测试或平台设施,未发裸 HTTP,沿用 `.env` 现有超时与单源配置。 + +### 8.5 提交前收口门 + +| 门 | 证据 | 结果 | +| --- | --- | --- | +| `make check` | `release/check.log/.exit` | **exit 0**:94 文件格式通过、ruff 通过、import-linter **1 kept / 0 broken** | +| 包单测(两处版本一致) | `release/package.log/.exit` | **6 passed,exit 0** | +| 全量 unit | `release/unit.log/.exit` | **1469 passed,exit 0** | + +23 项 skip 的理由已逐条留档(`final-gates/skip-reasons.log`):17 项 redis 时间语义由 integration 变体覆盖、6 项需实验室语料 `data/soak/chs_images`;**skip 不计通过**。conda 启动器既有 `RequestsDependencyWarning` 保留,不宣称零告警。 + +### 8.6 尚未执行(不得当成已完成) + +merge、push、`git tag -a v1.3.5`、`python -m build`、`twine check`/`upload`、`pip download` 解包验证、Gitea Release 与包页面/仓库关联检查**全部未执行**;`pytest -m slow` 的其余 e2e(本轮只跑了 `test_smoke_gateway.py` 四节点)与 live 覆盖基线同样未跑。§6 的四项缺口除"独立验证"已由父会话关闭外,其余保持开启。 diff --git a/src/polygateway/__init__.py b/src/polygateway/__init__.py index 4043c5a..52242e0 100644 --- a/src/polygateway/__init__.py +++ b/src/polygateway/__init__.py @@ -51,7 +51,7 @@ from polygateway.types import ( ThinkingObservation, ) -__version__ = "1.3.4" +__version__ = "1.3.5" __all__ = [ "DEFAULT_PROFILES",