From cb2deda08a36de69f34e0b71f73bbb70576c5a0c Mon Sep 17 00:00:00 2001 From: lhl Date: Sun, 9 Aug 2026 09:15:36 +0800 Subject: [PATCH] =?UTF-8?q?test:=20=E6=B8=85=E7=90=86=20backlog=E2=80=94?= =?UTF-8?q?=E2=80=94chat=5Fstructured=20=E8=B6=85=E9=99=90=E6=A3=80?= =?UTF-8?q?=E6=B5=8B=20+=20client=20=E4=B8=8A=E4=B8=8B=E6=96=87=E7=AE=A1?= =?UTF-8?q?=E7=90=86=E5=99=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- _AI_USAGE_LOG.md | 1 + src/genesis/inference/client.py | 7 +++++++ src/genesis/inference/engine.py | 3 +++ tests/test_inference_client.py | 18 ++++++++++++++++++ tests/test_inference_engine.py | 32 +++++++++++++++++++++++++++----- 5 files changed, 56 insertions(+), 5 deletions(-) diff --git a/_AI_USAGE_LOG.md b/_AI_USAGE_LOG.md index 3aa22c4..906513a 100644 --- a/_AI_USAGE_LOG.md +++ b/_AI_USAGE_LOG.md @@ -50,3 +50,4 @@ | 2026-08-09 10:40 | 反馈迭代 | 里程碑3.1 最终全分支评审(base 2fbff07..8bc5e5e)修正:评审查到 client 两处缺陷并修复——① 2xx 但响应结构损坏(非 JSON/缺 choices 字段)从裸 KeyError/JSONDecodeError 逃逸出 engine 的 except LLMError,改为捕获 (json.JSONDecodeError, KeyError, IndexError, TypeError) 后抛 LLMResponseError(spec §3.8 既有类型首次被触发,封闭 api-design §7 LLM_PARSE_ERROR 来源列虚指);② 3xx 重定向被当作成功(httpx 不自动跟随),补显式 2xx 判定,非 2xx 一律 LLMNetworkError。TDD:新增 2 用例(test_chat_3xx_no_retry、test_chat_malformed_response_raises_llm_response_error 含 not-json/missing-key 两种 handler)先 RED→实现→GREEN;聚焦 9 passed;pytest 全量 117 passed 覆盖 100.00%(785 stmts/186 br),fail_under=99 达标 | src/genesis/inference/client.py, tests/test_inference_client.py, _AI_USAGE_LOG.md | deepseek-v4-flash-free | (End of file - total 50 lines) +| 2026-08-09 10:50 | 反馈迭代 | 里程碑3.1 backlog 清理(retro 3 项):① chat_structured 增加 token 超限检测(渲染后估算超限→truncate_cb→重渲染,与 chat 流程一致)TDD RED→GREEN 新增 test_chat_structured_truncation_callback_triggered;② 清理 test_inference_engine.py 未用导入(pytest/json/ChatMessage);③ HttpLLMClient 增加上下文管理器(__enter__/__exit__ 退出时关闭底层 httpx.Client 释放连接)TDD 新增 test_client_context_manager_closes_transport;pytest 全量 119 passed 覆盖 100.00%,fail_under=99 达标;提交见 git log | src/genesis/inference/engine.py, src/genesis/inference/client.py, tests/test_inference_engine.py, tests/test_inference_client.py, _AI_USAGE_LOG.md | deepseek-v4-flash-free | diff --git a/src/genesis/inference/client.py b/src/genesis/inference/client.py index 2ffca79..17aead4 100644 --- a/src/genesis/inference/client.py +++ b/src/genesis/inference/client.py @@ -48,6 +48,13 @@ class HttpLLMClient: self._retry_backoff = retry_backoff self._client = httpx.Client(timeout=timeout_sec, transport=transport) + def __enter__(self) -> HttpLLMClient: + """支持 with 块:退出时自动关闭底层连接。""" + return self + + def __exit__(self, exc_type, exc_val, exc_tb) -> None: + self._client.close() + def chat( self, *, diff --git a/src/genesis/inference/engine.py b/src/genesis/inference/engine.py index dd91c02..f8bae78 100644 --- a/src/genesis/inference/engine.py +++ b/src/genesis/inference/engine.py @@ -133,6 +133,9 @@ class InferenceEngine: retry_count: int = 2, ) -> StructuredResult: rendered = self._render_prompt(prompt, variables) + if self._estimator(rendered) > self._max_context_tokens: + variables = self._apply_truncation(rendered, variables) + rendered = self._render_prompt(prompt, variables) # 追加 schema 约束说明(不强制模板支持) schema_hint = json.dumps(schema, ensure_ascii=False) if schema else "" base_rendered = rendered + (f'\n\n请输出符合以下 JSON Schema 的 JSON:{schema_hint}' if schema_hint else "") diff --git a/tests/test_inference_client.py b/tests/test_inference_client.py index cca307e..3bb1e47 100644 --- a/tests/test_inference_client.py +++ b/tests/test_inference_client.py @@ -54,6 +54,24 @@ def test_chat_requires_api_key(): HttpLLMClient(base_url="https://x", api_key="") +def test_client_context_manager_closes_transport(): + # with 块退出后底层 httpx.Client 应被关闭(释放连接) + with HttpLLMClient( + base_url="https://api.test.local", + api_key="sk-test", + transport=httpx.MockTransport(_ok_handler), + ) as client: + assert isinstance(client, HttpLLMClient) + text, _ = client.chat( + model="deepseek-chat", + messages=[ChatMessage(role="user", content="hi")], + temperature=0.2, + max_tokens=100, + ) + assert text == "Hello" + assert client._client.is_closed is True + + def test_chat_timeout(): def slow(request): raise httpx.ReadTimeout("slow") diff --git a/tests/test_inference_engine.py b/tests/test_inference_engine.py index 3d07704..93eb8e5 100644 --- a/tests/test_inference_engine.py +++ b/tests/test_inference_engine.py @@ -1,14 +1,10 @@ from __future__ import annotations -import json - -import pytest - from genesis.inference.engine import InferenceEngine from genesis.inference.exceptions import LLMError from genesis.inference.prompt_registry import PromptRegistry from genesis.inference.token import approximate_token_count -from genesis.inference.types import ChatMessage, Prompt +from genesis.inference.types import Prompt from tests.inference_helpers import FakeLLMClient @@ -161,6 +157,32 @@ def test_chat_structured_failed_on_network(): assert r.status == "failed" +def test_chat_structured_truncation_callback_triggered(): + # chat_structured 超限时同样触发 truncate_cb(与 chat 流程一致) + seen = {} + + def truncate_cb(prompt_text, variables): + seen["called"] = True + return {**variables, "chapter": "裁剪版"} + + eng = InferenceEngine( + client=FakeLLMClient([("ok", '{"a": 1}')]), + models=Models(), + registry=PromptRegistry(), + estimator=approximate_token_count, + truncate_cb=truncate_cb, + ) + eng._max_context_tokens = 2 # 强制超限(渲染约 3 token > 2) + r = eng.chat_structured( + session_id="s1", + prompt=Prompt(name="p", version="v1", template="abcd{{ chapter }}"), + variables={"chapter": "很长很长的标题"}, + schema={"type": "object", "properties": {"a": {"type": "number"}}}, + ) + assert seen["called"] is True + assert r.status == "ok" + + # ---------- 补充分支覆盖(defensive / 缺失配置) ---------- def test_chat_fallback_all_failed_when_only_primary():