diff --git a/Makefile b/Makefile index 5ebf1e7..5f3bf3b 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: install test lint format check ci wiki +.PHONY: install test lint format check ci wiki wiki-check ENV := PolyGateway @@ -24,3 +24,10 @@ ci: check test wiki: conda run -n $(ENV) python3 .claude/tools/research_wiki.py rebuild_index research-wiki/ + +# 用户文档站(Gitea Wiki)与源码的机械对齐校验。wiki 是独立仓库,须显式给路径: +# make wiki-check WIKI=~/PolyGateway.wiki +# 不并入 ci: 仓库里没有 wiki,自动跳过等于静默降级(违 P5),宁可让人显式跑。 +wiki-check: + @test -n "$(WIKI)" || (echo "用法: make wiki-check WIKI=" && exit 1) + conda run -n $(ENV) python3 tools/check_wiki_alignment.py --wiki $(WIKI) diff --git a/tools/check_wiki_alignment.py b/tools/check_wiki_alignment.py new file mode 100644 index 0000000..2a7857f --- /dev/null +++ b/tools/check_wiki_alignment.py @@ -0,0 +1,220 @@ +#!/usr/bin/env python3 +"""机械校验 Gitea Wiki 与源码的可比对事实(签名/导出/字段序/列清单/env 键)。 + +**只查机械可比对的部分**——机制语义、行为口径这类需要读懂代码才能判断的断言 +不在此列(那部分靠 `解释-治理行为` 的适用性总表做单一事实源 + 人工审查)。 + +设计动机: 2026-08 对 wiki 做了七轮人工审查,88 条发现里有相当一部分属于 +"机械可校验却写错"——`gather_bounded(coros, limit)`(实为 keyword-only 的 +`concurrency`)、`LLMResponse` 字段表把 `source_name` 排进前 11 位、遥测列数 +写成 20/21(实为 22 列)、`__version__` 在自称"全集"的页面缺席。这类偏差不该 +靠人一轮轮追,故收敛为脚本。 + +用法(wiki 是独立仓库,须显式给路径;**不做 skip 静默降级**): + python3 tools/check_wiki_alignment.py --wiki /path/to/PolyGateway.wiki +""" + +from __future__ import annotations + +import argparse +import dataclasses +import inspect +import re +import sys +from pathlib import Path + +import polygateway +from polygateway import EmbeddingClient, GatewayClient, LLMResponse +from polygateway.config import _SOURCE_FIELDS +from polygateway.ocr import OcrClient +from polygateway.providers import register_provider +from polygateway.telemetry.sqlite import _COLUMNS as TELEMETRY_COLUMNS + +# 参数名允许在 wiki 里以别名出现的白名单(仅限确无歧义的自解释形参) +_PARAM_ALIASES: dict[str, set[str]] = {"env": {"env"}} + +# (符号, 可调用对象) —— 这些的签名必须在 wiki 里逐参数出现 +_SIGNATURE_TARGETS = [ + ("GatewayClient.chat", GatewayClient.chat), + ("GatewayClient.from_env", GatewayClient.from_env), + ("EmbeddingClient.from_env", EmbeddingClient.from_env), + ("EmbeddingClient.embed", EmbeddingClient.embed), + ("OcrClient.from_env", OcrClient.from_env), + ("OcrClient.recognize_text", OcrClient.recognize_text), + ("gather_bounded", polygateway.gather_bounded), + ("register_provider", register_provider), +] + + +def _wiki_text(wiki: Path) -> dict[str, str]: + """读全部 .md;文件名(不含后缀)→ 正文。""" + pages = {p.stem: p.read_text(encoding="utf-8") for p in sorted(wiki.glob("*.md"))} + if not pages: + raise SystemExit(f"错误: {wiki} 下没有 .md 文件,路径是否指向 wiki 克隆?") + return pages + + +def check_exports_documented(pages: dict[str, str]) -> list[str]: + """`__all__` 每一项都得在某页出现过(R1 漏 gather_bounded、R6 漏 __version__)。""" + blob = "\n".join(pages.values()) + missing = [name for name in polygateway.__all__ if name not in blob] + return [f"__all__ 的 {name!r} 在全部 wiki 页面中零命中(页首自称『顶层导出全集』)" + for name in missing] + + +def check_signatures(pages: dict[str, str]) -> list[str]: + """提到某个公共可调用的那一行,必须列全它的参数名。 + + 只查「参数名是否出现」,不查顺序与类型——后者用自然语言表述合法。 + 历史命中: gather_bounded 的 concurrency 被写成 limit;EmbeddingClient/ + OcrClient 的 from_env 用省略号承接 chat 的关键字集合,掩盖了没有 cache=。 + """ + problems = [] + for label, func in _SIGNATURE_TARGETS: + symbol = label.split(".")[-1] + params = [ + p.name + for p in inspect.signature(func).parameters.values() + if p.name not in ("self", "cls") + ] + # 找出提到该符号的所有行,任一行列全即算通过 + lines = [ + line + for text in pages.values() + for line in text.splitlines() + if f"`{symbol}`" in line or f"{symbol}(" in line + ] + if not lines: + problems.append(f"{label}: wiki 里找不到任何提及") + continue + best_missing: list[str] | None = None + for line in lines: + missing = [ + p for p in params + if p not in line and not (_PARAM_ALIASES.get(p, set()) & set(line.split())) + ] + if not missing: + best_missing = [] + break + if best_missing is None or len(missing) < len(best_missing): + best_missing = missing + if best_missing: + problems.append( + f"{label}: 没有任何一行列全参数,最接近的一行仍缺 {best_missing}" + f"(实际签名 {inspect.signature(func)})" + ) + return problems + + +def check_llmresponse_field_order(pages: dict[str, str]) -> list[str]: + """字段表出现顺序须与 dataclass 声明顺序一致。 + + R2 命中: wiki 把 source_name 与 model/provider 并成一行(位置 5),而它实为 + 第 12 个字段。迁移中的三项目按位置构造 fake,照 wiki 写会静默错位。 + """ + page = pages.get("参考-公共API") + if page is None: + return ["缺少 参考-公共API.md"] + # 只在 LLMResponse 小节内找: 别的类型(EmbeddingResponse 等)也有同名字段, + # 全页搜索会命中它们、把顺序判断带偏 + start = page.find("## LLMResponse") + if start < 0: + return ["参考-公共API.md 缺少 `## LLMResponse` 小节"] + end = page.find("\n## ", start + 1) + section = page[start : end if end > 0 else len(page)] + declared = [f.name for f in dataclasses.fields(LLMResponse)] + positions = [] + for name in declared: + # 取该字段在小节内最早的出现位置(表格首列可能写成 `a / b` 合并形式) + cands = [ + section.find(pat) + for pat in (f"| {name} ", f"{name} /", f"/ {name} ", f"| {name}\n") + ] + hits = [i for i in cands if i >= 0] + positions.append((name, min(hits) if hits else -1)) + documented = [n for n, i in positions if i >= 0] + missing = [n for n, i in positions if i < 0] + problems = [f"LLMResponse 字段 {missing} 未在 参考-公共API 的字段表出现"] if missing else [] + ordered = sorted((i, n) for n, i in positions if i >= 0) + actual = [n for _, n in ordered] + expected = [n for n in declared if n in documented] + if actual != expected: + problems.append( + f"LLMResponse 字段表顺序与声明顺序不符\n" + f" wiki 顺序: {actual}\n" + f" 声明顺序: {expected}" + ) + return problems + + +def check_telemetry_columns(pages: dict[str, str]) -> list[str]: + """遥测列清单与 sqlite 后端的 _COLUMNS 对齐(created_at 由 DDL 生成,单列)。""" + page = pages.get("指南-遥测与成本") + if page is None: + return ["缺少 指南-遥测与成本.md"] + expected = [*TELEMETRY_COLUMNS, "created_at"] + missing = [c for c in expected if c not in page] + problems = [f"遥测列 {missing} 未在 指南-遥测与成本 出现"] if missing else [] + # 列数声明: 表 = _COLUMNS + created_at;端口 = _COLUMNS。 + # 页内**所有** "N 列" 声明都必须等于真实列数——只查"正确值是否出现"会被 + # 漏改的旧数字骗过(它们同时存在时检查照样通过) + table_n, port_n = len(expected), len(TELEMETRY_COLUMNS) + declared_counts = {int(m) for m in re.findall(r"(\d+)\s*列", page)} + if not declared_counts: + problems.append(f"指南-遥测与成本 未声明表列数(应为 {table_n} 列)") + elif declared_counts != {table_n}: + wrong = sorted(declared_counts - {table_n}) + problems.append( + f"指南-遥测与成本 的列数声明 {wrong} 与实际 {table_n} 列不符" + f"(端口是 {port_n} 参数,两者差 created_at)" + ) + return problems + + +def check_source_env_fields(pages: dict[str, str]) -> list[str]: + """`_SOURCE_FIELDS` 的每个 FIELD 段都得在 参考-配置键 出现(R1 命中 EXTRA_BODY)。""" + page = pages.get("参考-配置键") + if page is None: + return ["缺少 参考-配置键.md"] + # 用词边界匹配: `f in page` 会让 EXTRA_BODY 被 EXTRA_BODYY 蒙混过关 + missing = [f for f in _SOURCE_FIELDS if not re.search(rf"\b{re.escape(f)}\b", page)] + return [f"源键 FIELD 段 {missing} 未在 参考-配置键 文档化"] if missing else [] + + +_CHECKS = [ + ("顶层导出覆盖", check_exports_documented), + ("公共签名参数", check_signatures), + ("LLMResponse 字段序", check_llmresponse_field_order), + ("遥测列清单", check_telemetry_columns), + ("源 env 键覆盖", check_source_env_fields), +] + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--wiki", required=True, type=Path, help="PolyGateway.wiki 克隆目录") + args = parser.parse_args() + if not args.wiki.is_dir(): + raise SystemExit(f"错误: {args.wiki} 不是目录") + + pages = _wiki_text(args.wiki) + failed = 0 + for label, check in _CHECKS: + problems = check(pages) + if problems: + failed += len(problems) + print(f"✗ {label}") + for p in problems: + print(f" {p}") + else: + print(f"✓ {label}") + print() + if failed: + print(f"{failed} 处机械偏差 —— wiki 与源码不一致") + return 1 + print(f"{len(pages)} 页机械校验通过") + return 0 + + +if __name__ == "__main__": + sys.exit(main())