diff --git a/src/polygateway/transports/monkey_ocr.py b/src/polygateway/transports/monkey_ocr.py index e07e990..9e0edc9 100644 --- a/src/polygateway/transports/monkey_ocr.py +++ b/src/polygateway/transports/monkey_ocr.py @@ -24,6 +24,11 @@ from polygateway.errors import ( SourceDeadError, TransientError, ) +from polygateway.transports._http_errors import ( + compose_message, + response_body, + summarize_body, +) from polygateway.types import ( OcrLayoutElement, OcrLayoutTransportResult, @@ -74,13 +79,20 @@ def _translate_http_errors(source_name: str, operation: str) -> Iterator[None]: def _classify_status( exc: httpx.HTTPStatusError, source_name: str, operation: str ) -> TransientError | SourceDeadError | RequestRejectedError: + """HTTP 状态码 → 错误四分类,**全部分支**携带响应体摘要(issue #10)。 + + 分类映射本身零变更;摘要口径与 chat 侧共用同一实现,不得在此另起一份—— + "只有一个分支用了响应体"正是 issue #10 的成因。 + """ status = exc.response.status_code + summary = summarize_body(response_body(exc.response)) ctx: dict[str, Any] = { "source_name": source_name, "status_code": status, "operation": operation, + "body_text": summary, } - message = f"{source_name} OCR {operation} HTTP {status}" + message = compose_message(f"{source_name} OCR {operation} HTTP {status}", summary) if status >= 500 or status == 429: return TransientError(message, **ctx) if status in (401, 403): diff --git a/tests/unit/test_monkey_ocr.py b/tests/unit/test_monkey_ocr.py index 98d59a7..87f052c 100644 --- a/tests/unit/test_monkey_ocr.py +++ b/tests/unit/test_monkey_ocr.py @@ -19,7 +19,11 @@ from polygateway.errors import ( SourceDeadError, TransientError, ) -from polygateway.transports.monkey_ocr import MonkeyOcrTransport, _parse_middle_json +from polygateway.transports.monkey_ocr import ( + MonkeyOcrTransport, + _classify_status, + _parse_middle_json, +) from polygateway.types import SourceConfig @@ -306,6 +310,40 @@ class TestErrorTranslation: await t.recognize_text(image=b"jpg", source=_source(), call_id="c1") assert ei.value.status_code == status + @pytest.mark.parametrize( + ("status", "exc_type"), + [(502, TransientError), (401, SourceDeadError), (404, RequestRejectedError)], + ) + async def test_body_survives_every_branch(self, status, exc_type): + """issue #10: OCR 侧 message 原本只有 HTTP 状态码,拒绝理由同样丢失。""" + body = '{"detail":"unsupported image mode CMYK"}' + t = _transport_for(_routes(text_resp=httpx.Response(status, content=body.encode()))) + with pytest.raises(exc_type) as ei: + await t.recognize_text(image=b"jpg", source=_source(), call_id="c1") + assert ei.value.body_text == body + assert str(ei.value).endswith(f" | {body}") + + def test_unread_body_degrades_without_changing_class(self): + """取不到 body 时降级空串: 绝不能让 ResponseNotRead 逃出错误四分类。 + + 直接测纯函数而非走 MockTransport——真实客户端对非 stream 请求总会读完 + 响应,未读态只可能在将来给 OCR 加 stream 时出现,而那正是要防的场景。 + """ + + class _Unread(httpx.SyncByteStream): + def __iter__(self): + yield b"body" + + exc = httpx.HTTPStatusError( + "404", + request=httpx.Request("POST", "http://ocr.example/ocr/text"), + response=httpx.Response(404, stream=_Unread()), + ) + err = _classify_status(exc, "monkey_1", "text") + assert isinstance(err, RequestRejectedError) + assert err.body_text == "" + assert str(err) == "monkey_1 OCR text HTTP 404" + async def test_connect_error_transient(self): def handler(request): raise httpx.ConnectError("refused", request=request)